officeparser 7.0.2 → 7.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +61 -2
- package/dist/OfficeConverter.d.ts +1 -1
- package/dist/OfficeParser.d.ts +1 -1
- package/dist/OfficeParser.js +9 -0
- package/dist/defaults.js +16 -1
- package/dist/generators/ChunkingGenerator.js +23 -3
- package/dist/generators/PdfGenerator.js +51 -4
- package/dist/generators/RtfGenerator.js +36 -17
- package/dist/officeparser.browser.d.ts +121 -4
- package/dist/officeparser.browser.iife.js +57 -53
- package/dist/officeparser.browser.mjs +57 -53
- package/dist/parsers/CsvParser.js +5 -0
- package/dist/parsers/ExcelParser.js +6 -2
- package/dist/parsers/HtmlParser.js +5 -0
- package/dist/parsers/MarkdownParser.js +5 -0
- package/dist/parsers/OpenOfficeParser.js +4 -0
- package/dist/parsers/PdfParser.js +3 -0
- package/dist/parsers/PowerPointParser.js +4 -0
- package/dist/parsers/RtfParser.js +2 -0
- package/dist/parsers/WordParser.js +4 -0
- package/dist/sbom.cdx.json +99 -99
- package/dist/types.d.ts +119 -2
- package/dist/types.js +3 -1
- package/dist/utils/configUtils.js +14 -1
- package/dist/utils/errorUtils.d.ts +20 -0
- package/dist/utils/errorUtils.js +37 -2
- package/dist/utils/moduleLoader.js +3 -3
- package/dist/utils/ocrUtils.js +271 -66
- package/dist/utils/xmlUtils.d.ts +8 -0
- package/dist/utils/xmlUtils.js +33 -1
- package/package.json +3 -2
package/README.md
CHANGED
|
@@ -178,6 +178,60 @@ const ast = await officeParser.parseOffice(buffer);
|
|
|
178
178
|
> const ast = await officeParser.parseOffice(markdownBuffer, { fileType: 'md' });
|
|
179
179
|
> ```
|
|
180
180
|
|
|
181
|
+
### Cancellation with AbortSignal
|
|
182
|
+
|
|
183
|
+
You can pass a standard `AbortSignal` (e.g. from an `AbortController`) to cancel an active parse operation. This is especially useful for setting request-level timeouts or canceling long-running parses (like large PDFs with OCR).
|
|
184
|
+
|
|
185
|
+
```js
|
|
186
|
+
const controller = new AbortController();
|
|
187
|
+
|
|
188
|
+
// Cancel parsing if it takes longer than 5 seconds
|
|
189
|
+
setTimeout(() => controller.abort(), 5000);
|
|
190
|
+
|
|
191
|
+
try {
|
|
192
|
+
const ast = await officeParser.parseOffice('large_scanned_file.pdf', {
|
|
193
|
+
abortSignal: controller.signal,
|
|
194
|
+
ocr: true
|
|
195
|
+
});
|
|
196
|
+
} catch (err) {
|
|
197
|
+
if (err.name === 'AbortError') {
|
|
198
|
+
console.log('Parsing was cancelled.');
|
|
199
|
+
} else {
|
|
200
|
+
console.error('Parsing failed:', err);
|
|
201
|
+
}
|
|
202
|
+
}
|
|
203
|
+
```
|
|
204
|
+
|
|
205
|
+
> [!IMPORTANT]
|
|
206
|
+
> **AbortError Propagation**
|
|
207
|
+
> When parsing is cancelled via `AbortSignal`, the parser rejects with a standard `AbortError` (a `DOMException` or an Error with `name: 'AbortError'`).
|
|
208
|
+
> This error is *not* wrapped in standard OfficeParser error types so that you can reliably detect cancellation using `error.name === 'AbortError'`.
|
|
209
|
+
|
|
210
|
+
> [!NOTE]
|
|
211
|
+
> **Worker Cleanup on Abort**
|
|
212
|
+
> If an OCR job is actively running in the background when the signal is aborted, `officeParser` automatically terminates the Tesseract worker process immediately and removes it from the pool to prevent thread/memory leaks.
|
|
213
|
+
|
|
214
|
+
### Custom OCR Timeouts
|
|
215
|
+
|
|
216
|
+
To prevent the parser from hanging indefinitely due to slow network connections (when downloading Tesseract language datasets) or complex image processing, you can configure granular timeouts under `ocrConfig.timeout`.
|
|
217
|
+
|
|
218
|
+
```js
|
|
219
|
+
const ast = await officeParser.parseOffice('scanned_document.pdf', {
|
|
220
|
+
ocr: true,
|
|
221
|
+
ocrConfig: {
|
|
222
|
+
timeout: {
|
|
223
|
+
workerLoad: 30000, // 30s max to load worker & download language training files
|
|
224
|
+
recognition: 15000, // 15s max per image text recognition
|
|
225
|
+
autoTerminate: 10000 // 10s of inactivity before terminating idle workers
|
|
226
|
+
}
|
|
227
|
+
}
|
|
228
|
+
});
|
|
229
|
+
```
|
|
230
|
+
|
|
231
|
+
> [!TIP]
|
|
232
|
+
> **Non-Fatal Timeout Recovery**
|
|
233
|
+
> If `workerLoad` or `recognition` timeouts are exceeded, the parser will log a warning in `ast.warnings` and **continue parsing the rest of the document**. The overall promise resolves successfully with the text extracted from the document layers (rather than failing the entire parse).
|
|
234
|
+
|
|
181
235
|
### `ast.to()` — Generate from AST
|
|
182
236
|
|
|
183
237
|
The preferred way to convert a parsed AST to another format. Returns a `ConversionResult`.
|
|
@@ -606,10 +660,11 @@ Pass as the second argument to `parseOffice(file, config)`.
|
|
|
606
660
|
| `preserveXmlWhitespace` | `boolean` | `false` | Preserve original XML whitespace during serialization |
|
|
607
661
|
| `includeBreakNodes` | `boolean` | `false` | Include `w:br` / `w:cr` as typed break nodes (DOCX only) |
|
|
608
662
|
| `ignoreInternalLinks` | `boolean` | `false` | Strip bookmarks and internal cross-references from AST |
|
|
609
|
-
| `fileType` | `SupportedFileType \| null` | `null` | **Required for text-based
|
|
663
|
+
| `fileType` | `SupportedFileType \| null` | `null` | **Required for text-based binary data** (`'md'`, `'html'`, `'csv'`) as these lack magic bytes. |
|
|
610
664
|
| `csvDelimiter` | `string` | `','` | Input delimiter when parsing CSV files |
|
|
611
665
|
| `pdfWorkerSrc` | `string` | CDN (jsDelivr) | Path/URL to `pdf.worker.min.mjs` (required in browser) |
|
|
612
666
|
| `onWarning` | `(issue: OfficeIssue) => void` | — | Callback for non-fatal parsing issues |
|
|
667
|
+
| `abortSignal` | `AbortSignal \| null` | `null` | Optional signal to cancel parsing (rejects with AbortError) |
|
|
613
668
|
| `outputErrorToConsole` | `boolean` | `false` | **Deprecated.** Use `onWarning` instead |
|
|
614
669
|
|
|
615
670
|
---
|
|
@@ -630,6 +685,7 @@ Options shared by all generator formats. Pass to `OfficeGenerator.generate(ast,
|
|
|
630
685
|
| `styleMap` | `string[] \| StructuredStyleMapping[]` | `[]` | Custom semantic style mappings |
|
|
631
686
|
| `onNode` | `(node) => string \| false \| void` | — | Per-node callback for filtering, overriding, or mutating |
|
|
632
687
|
| `onWarning` | `(issue: OfficeIssue) => void` | — | Callback for non-fatal generation issues |
|
|
688
|
+
| `abortSignal` | `AbortSignal \| null` | `null` | Optional signal to cancel the generation operation (rejects with AbortError) |
|
|
633
689
|
|
|
634
690
|
---
|
|
635
691
|
|
|
@@ -732,6 +788,7 @@ Pass as `pdfConfig` inside `GeneratorConfig`. Requires the optional `puppeteer`
|
|
|
732
788
|
| `footerTemplate` | `string` | `''` | HTML template for the print footer |
|
|
733
789
|
| `scale` | `number` | `1` | Rendering scale factor |
|
|
734
790
|
| `launchOptions` | `object` | headless defaults | Puppeteer launch options (e.g., `executablePath`) |
|
|
791
|
+
| `timeout` | `number` | `30000` | PDF rendering timeout in milliseconds. Set to `0` to disable. |
|
|
735
792
|
|
|
736
793
|
### CsvGeneratorConfig
|
|
737
794
|
|
|
@@ -807,6 +864,7 @@ Configuration for `OfficeConverter.convert(file, format, config)`.
|
|
|
807
864
|
| `maxChunkSize` | `number` | `2000` | Max characters even if similarity stays high |
|
|
808
865
|
| `bufferSize` | `number` | `1` | Surrounding sentences used when computing similarity |
|
|
809
866
|
| `embeddingBatchSize` | `number` | `50` | Sentences per embedding API batch |
|
|
867
|
+
| `timeout` | `number` | `10000` | Timeout in milliseconds for individual embedding API calls. Set to `0` to disable. |
|
|
810
868
|
|
|
811
869
|
---
|
|
812
870
|
|
|
@@ -826,7 +884,8 @@ When `ocr: true` is set, `officeParser` maintains an intelligent **Smart Worker
|
|
|
826
884
|
| `workerPath` | `string` | `''` | Custom path to Tesseract worker script |
|
|
827
885
|
| `corePath` | `string` | `''` | Custom path to Tesseract core script |
|
|
828
886
|
| `langPath` | `string` | `''` | Custom path for language data files |
|
|
829
|
-
| `
|
|
887
|
+
| `timeout` | `OcrTimeoutConfig` | `{}` | Consolidated timeouts: `autoTerminate`, `workerLoad`, `recognition` |
|
|
888
|
+
| `autoTerminateTimeout` | `number` | `10000` | **Deprecated.** Use `timeout.autoTerminate` instead |
|
|
830
889
|
|
|
831
890
|
See all language codes at [tesseract-ocr.github.io](https://tesseract-ocr.github.io/tessdoc/Data-Files).
|
|
832
891
|
|
|
@@ -42,6 +42,6 @@ export declare class OfficeConverter {
|
|
|
42
42
|
* });
|
|
43
43
|
* ```
|
|
44
44
|
*/
|
|
45
|
-
static convert<F extends string | Buffer | ArrayBuffer, T extends SupportedFileType = InferFileTypeFromPath<F>>(file: F, destination: SupportedDestination<T>, config?: OfficeConverterConfig<SupportedDestination<T>, T>): Promise<ConversionResult<SupportedDestination<T>>>;
|
|
45
|
+
static convert<F extends string | Buffer | ArrayBuffer | Uint8Array, T extends SupportedFileType = InferFileTypeFromPath<F>>(file: F, destination: SupportedDestination<T>, config?: OfficeConverterConfig<SupportedDestination<T>, T>): Promise<ConversionResult<SupportedDestination<T>>>;
|
|
46
46
|
}
|
|
47
47
|
export {};
|
package/dist/OfficeParser.d.ts
CHANGED
|
@@ -91,7 +91,7 @@ export declare class OfficeParser {
|
|
|
91
91
|
* const text = ast.toText();
|
|
92
92
|
* ```
|
|
93
93
|
*/
|
|
94
|
-
static parseOffice(file: string | Buffer | ArrayBuffer, configOrCallback?: OfficeParserConfig | ((ast: OfficeParserAST, err?: any) => void), config?: OfficeParserConfig): Promise<OfficeParserAST>;
|
|
94
|
+
static parseOffice(file: string | Buffer | ArrayBuffer | Uint8Array, configOrCallback?: OfficeParserConfig | ((ast: OfficeParserAST, err?: any) => void), config?: OfficeParserConfig): Promise<OfficeParserAST>;
|
|
95
95
|
/**
|
|
96
96
|
* Terminates all active OCR workers and cleans up resources.
|
|
97
97
|
*
|
package/dist/OfficeParser.js
CHANGED
|
@@ -136,6 +136,9 @@ class OfficeParser {
|
|
|
136
136
|
if (file instanceof ArrayBuffer) {
|
|
137
137
|
buffer = Buffer.from(file);
|
|
138
138
|
}
|
|
139
|
+
else if (file instanceof Uint8Array) {
|
|
140
|
+
buffer = Buffer.from(file.buffer, file.byteOffset, file.byteLength);
|
|
141
|
+
}
|
|
139
142
|
else if (Buffer.isBuffer(file)) {
|
|
140
143
|
buffer = file;
|
|
141
144
|
}
|
|
@@ -238,6 +241,12 @@ class OfficeParser {
|
|
|
238
241
|
return result;
|
|
239
242
|
}
|
|
240
243
|
catch (error) {
|
|
244
|
+
// AbortError must pass through untouched so callers can distinguish a
|
|
245
|
+
// deliberate cancellation (err.name === 'AbortError') from a real parse failure.
|
|
246
|
+
// getWrappedError always creates a plain new Error(), which would strip the
|
|
247
|
+
// AbortError identity and break any instanceof / name checks on the caller side.
|
|
248
|
+
if (error?.name === 'AbortError')
|
|
249
|
+
throw error;
|
|
241
250
|
const wrappedError = (0, errorUtils_js_1.getWrappedError)(error, internalConfig, filePath);
|
|
242
251
|
if (callback)
|
|
243
252
|
callback(undefined, wrappedError);
|
package/dist/defaults.js
CHANGED
|
@@ -13,6 +13,12 @@ exports.DEFAULT_SENTENCE_BOUNDARY_REGEX = /[.!?。!?]/;
|
|
|
13
13
|
* Common abbreviations that should not trigger a sentence split when followed by a period.
|
|
14
14
|
*/
|
|
15
15
|
exports.DEFAULT_ABBREVIATIONS = ['Mr', 'Dr', 'Ms', 'Inc', 'Ltd', 'Prof', 'Sr', 'Jr', 'vs', 'etc'];
|
|
16
|
+
/** Default timeout values for OCR */
|
|
17
|
+
const DEFAULT_OCR_TIMEOUT = {
|
|
18
|
+
autoTerminate: 10000,
|
|
19
|
+
workerLoad: 60000,
|
|
20
|
+
recognition: 30000,
|
|
21
|
+
};
|
|
16
22
|
/**
|
|
17
23
|
* Default configuration for OCR.
|
|
18
24
|
*/
|
|
@@ -21,7 +27,12 @@ const DEFAULT_OCR_CONFIG = {
|
|
|
21
27
|
workerPath: '',
|
|
22
28
|
corePath: '',
|
|
23
29
|
langPath: '',
|
|
24
|
-
|
|
30
|
+
// Preferred: consolidated timeout object. New code should always read from here.
|
|
31
|
+
timeout: DEFAULT_OCR_TIMEOUT,
|
|
32
|
+
// Kept for backward compatibility. When timeout.autoTerminate is set (as above),
|
|
33
|
+
// the ocrUtils resolution logic will prefer timeout.autoTerminate over this flat field.
|
|
34
|
+
autoTerminateTimeout: DEFAULT_OCR_TIMEOUT.autoTerminate,
|
|
35
|
+
abortSignal: null,
|
|
25
36
|
};
|
|
26
37
|
/**
|
|
27
38
|
* Default configuration for the OfficeParser.
|
|
@@ -37,6 +48,7 @@ exports.DEFAULT_OFFICE_PARSER_CONFIG = {
|
|
|
37
48
|
ocr: false,
|
|
38
49
|
ocrLanguage: 'eng',
|
|
39
50
|
ocrConfig: DEFAULT_OCR_CONFIG,
|
|
51
|
+
abortSignal: null,
|
|
40
52
|
serializeRawContent: true,
|
|
41
53
|
preserveXmlWhitespace: false,
|
|
42
54
|
pdfWorkerSrc: DEFAULT_PDF_WORKER_SRC,
|
|
@@ -75,6 +87,7 @@ const DEFAULT_PDF_GENERATOR_CONFIG = {
|
|
|
75
87
|
headless: true,
|
|
76
88
|
args: ['--no-sandbox', '--disable-setuid-sandbox']
|
|
77
89
|
},
|
|
90
|
+
timeout: 30000,
|
|
78
91
|
};
|
|
79
92
|
/**
|
|
80
93
|
* Default configuration for CSV generation.
|
|
@@ -143,6 +156,7 @@ exports.DEFAULT_SEMANTIC_CHUNKING_CONFIG = {
|
|
|
143
156
|
lengthFunction: (text) => text.length,
|
|
144
157
|
sentenceBoundaryRegex: exports.DEFAULT_SENTENCE_BOUNDARY_REGEX,
|
|
145
158
|
abbreviations: exports.DEFAULT_ABBREVIATIONS,
|
|
159
|
+
timeout: 10000,
|
|
146
160
|
};
|
|
147
161
|
/**
|
|
148
162
|
* The resolved default chunking config (uses document-structure as default strategy).
|
|
@@ -162,6 +176,7 @@ exports.DEFAULT_GENERATOR_CONFIG = {
|
|
|
162
176
|
includeImages: true,
|
|
163
177
|
includeCharts: true,
|
|
164
178
|
ignoreInternalLinks: false,
|
|
179
|
+
abortSignal: null,
|
|
165
180
|
htmlConfig: DEFAULT_HTML_GENERATOR_CONFIG,
|
|
166
181
|
mdConfig: DEFAULT_MD_GENERATOR_CONFIG,
|
|
167
182
|
pdfConfig: DEFAULT_PDF_GENERATOR_CONFIG,
|
|
@@ -39,6 +39,7 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
39
39
|
* Note: ConversionResult.value is a JSON string of OfficeChunk[] for the 'chunks' destination.
|
|
40
40
|
*/
|
|
41
41
|
async generate() {
|
|
42
|
+
(0, errorUtils_js_1.checkAbortSignal)(this.config.abortSignal);
|
|
42
43
|
let chunks;
|
|
43
44
|
switch (this.chunkConfig.strategy) {
|
|
44
45
|
case 'fixed-size':
|
|
@@ -186,6 +187,7 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
186
187
|
return this.finalizeChunks(chunks, config);
|
|
187
188
|
}
|
|
188
189
|
async processNodeForStructure(node, config, splitBy, maxChunkSize, measure, chunks, contextStack) {
|
|
190
|
+
(0, errorUtils_js_1.checkAbortSignal)(this.config.abortSignal);
|
|
189
191
|
// Check for node override or skip
|
|
190
192
|
const override = await this.handleOnNode(node);
|
|
191
193
|
if (override === false)
|
|
@@ -412,7 +414,7 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
412
414
|
if (sentences.length === 0)
|
|
413
415
|
return [];
|
|
414
416
|
// Embed all sentences in batches to avoid rate limiting
|
|
415
|
-
const embeddings = await this.batchEmbeddings(sentences, config.embeddingFunction, batchSize);
|
|
417
|
+
const embeddings = await this.batchEmbeddings(sentences, config.embeddingFunction, batchSize, config.timeout);
|
|
416
418
|
// Calculate cosine similarity between adjacent sentence windows
|
|
417
419
|
const chunks = [];
|
|
418
420
|
let currentSentences = [];
|
|
@@ -479,6 +481,7 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
479
481
|
let currentPage;
|
|
480
482
|
let currentSheet;
|
|
481
483
|
const walk = async (node) => {
|
|
484
|
+
(0, errorUtils_js_1.checkAbortSignal)(this.config.abortSignal);
|
|
482
485
|
const override = await this.handleOnNode(node);
|
|
483
486
|
if (override === false)
|
|
484
487
|
return;
|
|
@@ -533,6 +536,7 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
533
536
|
let currentPage;
|
|
534
537
|
let currentSheet;
|
|
535
538
|
const walk = async (node) => {
|
|
539
|
+
(0, errorUtils_js_1.checkAbortSignal)(this.config.abortSignal);
|
|
536
540
|
const override = await this.handleOnNode(node);
|
|
537
541
|
if (override === false)
|
|
538
542
|
return;
|
|
@@ -602,11 +606,27 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
602
606
|
/**
|
|
603
607
|
* Helper to process embeddings in sequential batches to avoid API rate limits and memory issues.
|
|
604
608
|
*/
|
|
605
|
-
async batchEmbeddings(sentences, embedFn, batchSize = 50) {
|
|
609
|
+
async batchEmbeddings(sentences, embedFn, batchSize = 50, timeoutMs) {
|
|
606
610
|
const results = [];
|
|
607
611
|
for (let i = 0; i < sentences.length; i += batchSize) {
|
|
612
|
+
(0, errorUtils_js_1.checkAbortSignal)(this.config.abortSignal);
|
|
608
613
|
const batch = sentences.slice(i, i + batchSize);
|
|
609
|
-
const
|
|
614
|
+
const batchPromises = batch.map(s => {
|
|
615
|
+
const call = embedFn(s.text);
|
|
616
|
+
if (timeoutMs !== undefined && timeoutMs > 0) {
|
|
617
|
+
let timerId;
|
|
618
|
+
const timeoutPromise = new Promise((_, reject) => {
|
|
619
|
+
timerId = setTimeout(() => {
|
|
620
|
+
reject(new Error(`Embedding call timed out after ${timeoutMs}ms`));
|
|
621
|
+
}, timeoutMs);
|
|
622
|
+
});
|
|
623
|
+
return Promise.race([call, timeoutPromise]).finally(() => {
|
|
624
|
+
clearTimeout(timerId);
|
|
625
|
+
});
|
|
626
|
+
}
|
|
627
|
+
return call;
|
|
628
|
+
});
|
|
629
|
+
const batchResults = await Promise.all(batchPromises);
|
|
610
630
|
results.push(...batchResults);
|
|
611
631
|
}
|
|
612
632
|
return results;
|
|
@@ -3,6 +3,7 @@ Object.defineProperty(exports, "__esModule", { value: true });
|
|
|
3
3
|
exports.PdfGenerator = void 0;
|
|
4
4
|
const types_js_1 = require("../types.js");
|
|
5
5
|
const envUtils_js_1 = require("../utils/envUtils.js");
|
|
6
|
+
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
6
7
|
const BaseGenerator_js_1 = require("./BaseGenerator.js");
|
|
7
8
|
const HtmlGenerator_js_1 = require("./HtmlGenerator.js");
|
|
8
9
|
/**
|
|
@@ -44,6 +45,24 @@ class PdfGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
44
45
|
* Uses dynamic import to avoid bundling puppeteer into the library core.
|
|
45
46
|
*/
|
|
46
47
|
async generateInNode(html) {
|
|
48
|
+
const signal = this.config.abortSignal;
|
|
49
|
+
if (signal?.aborted) {
|
|
50
|
+
throw (0, errorUtils_js_1.getAbortError)();
|
|
51
|
+
}
|
|
52
|
+
let browser;
|
|
53
|
+
const onAbort = async () => {
|
|
54
|
+
if (browser) {
|
|
55
|
+
try {
|
|
56
|
+
await browser.close();
|
|
57
|
+
}
|
|
58
|
+
catch (e) {
|
|
59
|
+
// ignore
|
|
60
|
+
}
|
|
61
|
+
}
|
|
62
|
+
};
|
|
63
|
+
if (signal) {
|
|
64
|
+
signal.addEventListener('abort', onAbort);
|
|
65
|
+
}
|
|
47
66
|
try {
|
|
48
67
|
// Dynamic import for peer dependency
|
|
49
68
|
// @ts-ignore
|
|
@@ -70,11 +89,16 @@ class PdfGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
70
89
|
// to ensure transparency about the environment state. Programmatically fixing this
|
|
71
90
|
// would require force-downloading a ~300MB arm64 browser binary or switching to
|
|
72
91
|
// a system-installed Chrome, both of which are too intrusive for a library.
|
|
73
|
-
|
|
92
|
+
browser = await puppeteer.launch(launchOptions);
|
|
74
93
|
const page = await browser.newPage();
|
|
94
|
+
const pdfConfig = this.config.pdfConfig;
|
|
95
|
+
const timeout = pdfConfig.timeout;
|
|
96
|
+
if (timeout !== undefined && timeout > 0) {
|
|
97
|
+
page.setDefaultTimeout(timeout);
|
|
98
|
+
page.setDefaultNavigationTimeout(timeout);
|
|
99
|
+
}
|
|
75
100
|
// Set content and wait for network/assets to load
|
|
76
101
|
await page.setContent(html, { waitUntil: 'networkidle0' });
|
|
77
|
-
const pdfConfig = this.config.pdfConfig;
|
|
78
102
|
const pdfBuffer = await page.pdf({
|
|
79
103
|
format: pdfConfig.format,
|
|
80
104
|
width: pdfConfig.width,
|
|
@@ -88,18 +112,41 @@ class PdfGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
88
112
|
footerTemplate: pdfConfig.footerTemplate,
|
|
89
113
|
});
|
|
90
114
|
await browser.close();
|
|
115
|
+
browser = null;
|
|
91
116
|
return {
|
|
92
117
|
value: new Uint8Array(pdfBuffer),
|
|
93
118
|
messages: this.messages
|
|
94
119
|
};
|
|
95
120
|
}
|
|
96
121
|
catch (err) {
|
|
97
|
-
|
|
122
|
+
if (browser) {
|
|
123
|
+
try {
|
|
124
|
+
await browser.close();
|
|
125
|
+
}
|
|
126
|
+
catch (e) {
|
|
127
|
+
// ignore
|
|
128
|
+
}
|
|
129
|
+
browser = null;
|
|
130
|
+
}
|
|
131
|
+
if (signal?.aborted) {
|
|
132
|
+
throw (0, errorUtils_js_1.getAbortError)();
|
|
133
|
+
}
|
|
134
|
+
if (err.message && (err.message.includes('timeout') || err.message.includes('Timeout'))) {
|
|
135
|
+
this.warn(types_js_1.OfficeWarningType.PAGE_LOAD_FAILED, `PDF generation timed out: ${err.message}`);
|
|
136
|
+
}
|
|
137
|
+
else {
|
|
138
|
+
this.warn(types_js_1.OfficeWarningType.DEPENDENCY_LOAD_FAILED, `puppeteer. Please install it with 'npm install puppeteer'. Error: ${err.message}`);
|
|
139
|
+
}
|
|
98
140
|
return {
|
|
99
|
-
value:
|
|
141
|
+
value: new Uint8Array(),
|
|
100
142
|
messages: this.messages
|
|
101
143
|
};
|
|
102
144
|
}
|
|
145
|
+
finally {
|
|
146
|
+
if (signal) {
|
|
147
|
+
signal.removeEventListener('abort', onAbort);
|
|
148
|
+
}
|
|
149
|
+
}
|
|
103
150
|
}
|
|
104
151
|
/**
|
|
105
152
|
* Browser implementation using hidden iframe and native print.
|
|
@@ -13,7 +13,7 @@ class RtfGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
13
13
|
}
|
|
14
14
|
async generate() {
|
|
15
15
|
this.colorTable = [];
|
|
16
|
-
// We first process all nodes to collect colors
|
|
16
|
+
// We first process all nodes to collect colors and analyze structure
|
|
17
17
|
const bodyContent = await this.renderBody(this.ast);
|
|
18
18
|
let output = '{\\rtf1\\ansi\\deff0\n';
|
|
19
19
|
// 1. Info Group (Metadata)
|
|
@@ -61,7 +61,6 @@ class RtfGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
61
61
|
let body = '';
|
|
62
62
|
this.inTable = false;
|
|
63
63
|
const processor = async (node, childrenOutput) => {
|
|
64
|
-
// Handle Semantic Style Mapping for RTF using the semantic mapping helper
|
|
65
64
|
const mapping = this.getSemanticMapping(node);
|
|
66
65
|
if (mapping) {
|
|
67
66
|
if (mapping.tag === 'blockquote') {
|
|
@@ -112,7 +111,7 @@ class RtfGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
112
111
|
const pt = parseInt(f.size);
|
|
113
112
|
prefix += `\\fs${pt * 2} `;
|
|
114
113
|
}
|
|
115
|
-
text = `{\\
|
|
114
|
+
text = `{\\f0 ${prefix}${text}${suffix}}`;
|
|
116
115
|
}
|
|
117
116
|
if (meta?.link) {
|
|
118
117
|
const isInternal = meta.linkType !== 'external';
|
|
@@ -142,25 +141,16 @@ class RtfGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
142
141
|
else if (meta.alignment === 'justify')
|
|
143
142
|
pPr += '\\qj';
|
|
144
143
|
}
|
|
145
|
-
if (meta.paragraphIndentation) {
|
|
146
|
-
const ind = meta.paragraphIndentation;
|
|
147
|
-
if (ind.left)
|
|
148
|
-
pPr += `\\li${ind.left}`;
|
|
149
|
-
if (ind.right)
|
|
150
|
-
pPr += `\\ri${ind.right}`;
|
|
151
|
-
if (ind.firstLine)
|
|
152
|
-
pPr += `\\fi${ind.firstLine}`;
|
|
153
|
-
}
|
|
154
144
|
}
|
|
155
145
|
return `${pPr} ${childrenOutput}\\par\n`;
|
|
156
146
|
}
|
|
157
147
|
case 'list': {
|
|
158
148
|
const meta = node.metadata;
|
|
159
|
-
const
|
|
149
|
+
const level = meta?.indentation || 0;
|
|
150
|
+
const indent = (level + 1) * 360;
|
|
160
151
|
const isOrdered = meta?.listType === 'ordered';
|
|
161
152
|
const marker = isOrdered ? `${(meta.itemIndex ?? 0) + 1}. ` : '\\bullet ';
|
|
162
153
|
const listControl = isOrdered ? '\\pndec' : '\\pnbullet';
|
|
163
|
-
const level = meta?.indentation || 0;
|
|
164
154
|
const pPr = this.inTable ? '\\pard\\intbl' : '\\pard';
|
|
165
155
|
return `${pPr}\\li${indent}\\fi-360\\ilvl${level}${listControl} ${marker}${childrenOutput}\\par\n`;
|
|
166
156
|
}
|
|
@@ -168,11 +158,40 @@ class RtfGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
168
158
|
return `\\pard\\sa0\n${childrenOutput}`;
|
|
169
159
|
}
|
|
170
160
|
case 'row': {
|
|
171
|
-
|
|
161
|
+
const cells = node.children || [];
|
|
162
|
+
const pageWidth = 9000; // Standard twips width
|
|
163
|
+
const cellWidth = Math.floor(pageWidth / (cells.length || 1));
|
|
164
|
+
let cellDefs = '';
|
|
165
|
+
for (let i = 0; i < cells.length; i++) {
|
|
166
|
+
// Add basic cell borders and calculate width
|
|
167
|
+
cellDefs += `\\clbrdrt\\brdrs\\brdrw10\\clbrdrl\\brdrs\\brdrw10\\clbrdrb\\brdrs\\brdrw10\\clbrdrr\\brdrs\\brdrw10\\cellx${(i + 1) * cellWidth}`;
|
|
168
|
+
}
|
|
169
|
+
return `\\trowd\\trgaph108\\trleft-108${cellDefs}\n${childrenOutput}\\row\n`;
|
|
172
170
|
}
|
|
173
171
|
case 'cell': {
|
|
174
|
-
|
|
175
|
-
|
|
172
|
+
return `\\pard\\intbl\\sb60\\sa60 ${childrenOutput}\\cell\n`;
|
|
173
|
+
}
|
|
174
|
+
case 'image': {
|
|
175
|
+
if (!this.config.includeImages)
|
|
176
|
+
return '';
|
|
177
|
+
const meta = node.metadata;
|
|
178
|
+
const attachmentName = meta?.attachmentName;
|
|
179
|
+
const attachment = this.ast.attachments.find(a => a.name === attachmentName);
|
|
180
|
+
if (attachment && attachment.data) {
|
|
181
|
+
const type = attachment.extension === 'png' ? 'pngblip' : 'jpegblip';
|
|
182
|
+
// Convert base64 to hex
|
|
183
|
+
const binary = atob(attachment.data);
|
|
184
|
+
let hex = '';
|
|
185
|
+
for (let i = 0; i < binary.length; i++) {
|
|
186
|
+
const h = binary.charCodeAt(i).toString(16);
|
|
187
|
+
hex += h.length === 1 ? '0' + h : h;
|
|
188
|
+
if (i % 64 === 63)
|
|
189
|
+
hex += '\n'; // Add newlines for better RTF readability
|
|
190
|
+
}
|
|
191
|
+
// Default goals (approx 3 inches wide at 1440 twips per inch)
|
|
192
|
+
return `{\\pict\\${type}\\picwgoal4320\\pichgoal3240\n${hex}\n}\n`;
|
|
193
|
+
}
|
|
194
|
+
return '';
|
|
176
195
|
}
|
|
177
196
|
case 'break': {
|
|
178
197
|
return node.metadata?.breakType === 'page' ? '\\page\n' : '\\line\n';
|
|
@@ -17,7 +17,7 @@ export declare enum OfficeErrorType {
|
|
|
17
17
|
IMPROPER_ARGUMENTS = "IMPROPER_ARGUMENTS",
|
|
18
18
|
/** Error occurred while reading or processing file buffers */
|
|
19
19
|
IMPROPER_BUFFERS = "IMPROPER_BUFFERS",
|
|
20
|
-
/** Input type is not a supported type (string, Buffer, ArrayBuffer) */
|
|
20
|
+
/** Input type is not a supported type (string, Buffer, ArrayBuffer, Uint8Array) */
|
|
21
21
|
INVALID_INPUT = "INVALID_INPUT",
|
|
22
22
|
/** PDF worker source is missing (required in browser) */
|
|
23
23
|
PDF_WORKER_MISSING = "PDF_WORKER_MISSING",
|
|
@@ -30,7 +30,9 @@ export declare enum OfficeErrorType {
|
|
|
30
30
|
/** Output mapping in style mapping is invalid */
|
|
31
31
|
INVALID_OUTPUT_MAPPING = "INVALID_OUTPUT_MAPPING",
|
|
32
32
|
/** Semantic chunking strategy is selected but no embedding function is provided */
|
|
33
|
-
MISSING_EMBEDDING_FUNCTION = "MISSING_EMBEDDING_FUNCTION"
|
|
33
|
+
MISSING_EMBEDDING_FUNCTION = "MISSING_EMBEDDING_FUNCTION",
|
|
34
|
+
/** The operation was aborted */
|
|
35
|
+
OPERATION_ABORTED = "OPERATION_ABORTED"
|
|
34
36
|
}
|
|
35
37
|
/**
|
|
36
38
|
* Standard warning types for OfficeParser.
|
|
@@ -70,6 +72,68 @@ export declare enum OfficeWarningType {
|
|
|
70
72
|
/** A node was skipped because it only contained whitespace */
|
|
71
73
|
WHITESPACE_NODE_SKIPPED = "WHITESPACE_NODE_SKIPPED"
|
|
72
74
|
}
|
|
75
|
+
/**
|
|
76
|
+
* Consolidated timeout settings for OCR operations.
|
|
77
|
+
* Preferred over the individual flat timeout properties on {@link OcrConfig},
|
|
78
|
+
* which are now deprecated.
|
|
79
|
+
*
|
|
80
|
+
* If a key is present here, it takes priority over the corresponding deprecated
|
|
81
|
+
* flat property (e.g. `timeout.autoTerminate` wins over `autoTerminateTimeout`).
|
|
82
|
+
* Set any value to `0` to disable that specific timeout.
|
|
83
|
+
*/
|
|
84
|
+
export interface OcrTimeoutConfig {
|
|
85
|
+
/**
|
|
86
|
+
* Timeout in milliseconds of inactivity before the OCR worker pool is
|
|
87
|
+
* automatically terminated and freed.
|
|
88
|
+
*
|
|
89
|
+
* The timer resets every time a new OCR job is enqueued. When the last
|
|
90
|
+
* job completes and this duration passes without a new one, the entire
|
|
91
|
+
* worker pool is torn down so that no background threads keep the Node.js
|
|
92
|
+
* process alive unnecessarily.
|
|
93
|
+
*
|
|
94
|
+
* Set to `0` to keep workers alive indefinitely (useful when you want to
|
|
95
|
+
* call {@link terminateOcr} manually at shutdown time).
|
|
96
|
+
* Default is 10,000 ms (10 seconds).
|
|
97
|
+
*/
|
|
98
|
+
autoTerminate?: number;
|
|
99
|
+
/**
|
|
100
|
+
* Timeout in milliseconds for initializing a Tesseract worker
|
|
101
|
+
* (loading the JS runtime, downloading or loading the `.traineddata`
|
|
102
|
+
* language file) or for re-initializing an existing worker with a
|
|
103
|
+
* different language.
|
|
104
|
+
*
|
|
105
|
+
* Multi-language combinations (e.g. `'por+eng+spa'`) must download a
|
|
106
|
+
* separate `.traineddata` file for each language and are therefore
|
|
107
|
+
* particularly susceptible to slow networks. Tune this value upward if
|
|
108
|
+
* your OCR environment has high network latency or if you are loading
|
|
109
|
+
* languages from disk in a large container image.
|
|
110
|
+
*
|
|
111
|
+
* When the timeout fires, the failed job is rejected with a non-fatal
|
|
112
|
+
* {@link OfficeWarningType.OCR_FAILED} warning and parsing continues
|
|
113
|
+
* without OCR output for that image. The stalled worker is terminated
|
|
114
|
+
* and removed from the pool to prevent thread leaks.
|
|
115
|
+
*
|
|
116
|
+
* Set to `0` to wait indefinitely (not recommended for production; a hung
|
|
117
|
+
* network request will block the entire OCR queue for that language).
|
|
118
|
+
* Default is 60,000 ms (60 seconds).
|
|
119
|
+
*/
|
|
120
|
+
workerLoad?: number;
|
|
121
|
+
/**
|
|
122
|
+
* Timeout in milliseconds for the actual OCR text-recognition call
|
|
123
|
+
* (`worker.recognize(image)`) on an already-initialized Tesseract worker.
|
|
124
|
+
*
|
|
125
|
+
* Recognition time scales with image resolution and the number of active
|
|
126
|
+
* languages. Very high-resolution scans or unusual character sets can
|
|
127
|
+
* exceed the default. If this timeout fires, the job is rejected with a
|
|
128
|
+
* non-fatal {@link OfficeWarningType.OCR_FAILED} warning; the worker is
|
|
129
|
+
* terminated and evicted from the pool because its internal state after a
|
|
130
|
+
* mid-recognition timeout is undefined.
|
|
131
|
+
*
|
|
132
|
+
* Set to `0` to wait indefinitely.
|
|
133
|
+
* Default is 30,000 ms (30 seconds).
|
|
134
|
+
*/
|
|
135
|
+
recognition?: number;
|
|
136
|
+
}
|
|
73
137
|
/**
|
|
74
138
|
* Configuration options for OCR.
|
|
75
139
|
*/
|
|
@@ -104,11 +168,33 @@ export interface OcrConfig {
|
|
|
104
168
|
*/
|
|
105
169
|
langPath?: string;
|
|
106
170
|
/**
|
|
171
|
+
* Consolidated timeout settings for all OCR operations.
|
|
172
|
+
*
|
|
173
|
+
* Prefer this over the deprecated flat timeout properties.
|
|
174
|
+
* If `timeout.autoTerminate` is set, it takes priority over the deprecated `autoTerminateTimeout`.
|
|
175
|
+
*/
|
|
176
|
+
timeout?: OcrTimeoutConfig;
|
|
177
|
+
/**
|
|
178
|
+
* @deprecated Use `timeout.autoTerminate` instead.
|
|
179
|
+
*
|
|
107
180
|
* Timeout in milliseconds of inactivity before the OCR worker pool is automatically terminated.
|
|
108
181
|
* Set to 0 to disable auto-termination.
|
|
109
182
|
* Default is 10,000 (10 seconds).
|
|
183
|
+
*
|
|
184
|
+
* If `timeout.autoTerminate` is also set, that value takes priority over this one.
|
|
110
185
|
*/
|
|
111
186
|
autoTerminateTimeout?: number;
|
|
187
|
+
/**
|
|
188
|
+
* An optional AbortSignal propagated from the main parser configuration to abort active OCR jobs.
|
|
189
|
+
* If the signal is aborted:
|
|
190
|
+
* 1. Any pending OCR jobs in the scheduler queue are rejected immediately.
|
|
191
|
+
* 2. Any active OCR job running on a Tesseract worker will reject, the worker will be
|
|
192
|
+
* terminated, and it will be removed from the pool to avoid hanging worker threads.
|
|
193
|
+
*
|
|
194
|
+
* Developers should prefer passing this at the top level of `parseOffice` (as `config.abortSignal`),
|
|
195
|
+
* which automatically propagates here.
|
|
196
|
+
*/
|
|
197
|
+
abortSignal?: AbortSignal | null;
|
|
112
198
|
}
|
|
113
199
|
/**
|
|
114
200
|
* Configuration options for the OfficeParser.
|
|
@@ -175,6 +261,20 @@ export interface OfficeParserConfig {
|
|
|
175
261
|
* If provided, `ocrLanguage` will be ignored in favor of `ocrConfig.language`.
|
|
176
262
|
*/
|
|
177
263
|
ocrConfig?: OcrConfig;
|
|
264
|
+
/**
|
|
265
|
+
* An optional AbortSignal to cancel the parsing operation.
|
|
266
|
+
* When aborted, the parser immediately rejects with a standard AbortError (DOMException).
|
|
267
|
+
*
|
|
268
|
+
* ### Format-Specific Abort Behavior:
|
|
269
|
+
* - **PDF**: Checked between page loads and before individual image OCR operations.
|
|
270
|
+
* - **RTF**: Checked before parsing/traversal and before running OCR on image attachments.
|
|
271
|
+
* - **DOCX/XLSX/PPTX/ODF**: Checked during zip decompression before loading and parsing XML files.
|
|
272
|
+
* - **CSV/MD/HTML**: Checked at the start of the parsing phase.
|
|
273
|
+
*
|
|
274
|
+
* Note: If an OCR operation is currently running on a Tesseract worker when aborted,
|
|
275
|
+
* the worker will be terminated and removed from the worker pool automatically to prevent leaks.
|
|
276
|
+
*/
|
|
277
|
+
abortSignal?: AbortSignal | null;
|
|
178
278
|
/**
|
|
179
279
|
* Flag to serialize raw content (XML) as clean, formatted strings.
|
|
180
280
|
* Only relevant when `includeRawContent` is true.
|
|
@@ -363,6 +463,12 @@ export interface CommonGeneratorConfig {
|
|
|
363
463
|
* Defaults to false.
|
|
364
464
|
*/
|
|
365
465
|
ignoreInternalLinks?: boolean;
|
|
466
|
+
/**
|
|
467
|
+
* An optional AbortSignal to cancel the generation operation.
|
|
468
|
+
* When aborted, the generator immediately rejects with a standard AbortError.
|
|
469
|
+
* Currently supported by PdfGenerator and ChunkingGenerator.
|
|
470
|
+
*/
|
|
471
|
+
abortSignal?: AbortSignal | null;
|
|
366
472
|
}
|
|
367
473
|
/**
|
|
368
474
|
* Destination-aware generator configuration.
|
|
@@ -474,6 +580,12 @@ export interface PdfGeneratorConfig {
|
|
|
474
580
|
* Useful for setting custom executable paths or args in CI/CD.
|
|
475
581
|
*/
|
|
476
582
|
launchOptions?: any;
|
|
583
|
+
/**
|
|
584
|
+
* Timeout in milliseconds for PDF generation.
|
|
585
|
+
* Limits the time spent waiting for Puppeteer to launch, load content, and render PDF.
|
|
586
|
+
* Defaults to 30000 ms (30 seconds). Set to 0 to disable.
|
|
587
|
+
*/
|
|
588
|
+
timeout?: number;
|
|
477
589
|
}
|
|
478
590
|
/**
|
|
479
591
|
* Structured style mapping definition for the StyleMapper.
|
|
@@ -779,6 +891,11 @@ export interface SemanticChunkingConfig extends BaseChunkingConfig {
|
|
|
779
891
|
* Default is 50.
|
|
780
892
|
*/
|
|
781
893
|
embeddingBatchSize?: number;
|
|
894
|
+
/**
|
|
895
|
+
* Timeout in milliseconds for individual embedding API calls.
|
|
896
|
+
* Defaults to 10000 ms (10 seconds). Set to 0 to disable.
|
|
897
|
+
*/
|
|
898
|
+
timeout?: number;
|
|
782
899
|
}
|
|
783
900
|
/**
|
|
784
901
|
* Discriminated union of all chunking strategy configurations.
|
|
@@ -1546,7 +1663,7 @@ export declare class OfficeParser {
|
|
|
1546
1663
|
* const text = ast.toText();
|
|
1547
1664
|
* ```
|
|
1548
1665
|
*/
|
|
1549
|
-
static parseOffice(file: string | Buffer | ArrayBuffer, configOrCallback?: OfficeParserConfig | ((ast: OfficeParserAST, err?: any) => void), config?: OfficeParserConfig): Promise<OfficeParserAST>;
|
|
1666
|
+
static parseOffice(file: string | Buffer | ArrayBuffer | Uint8Array, configOrCallback?: OfficeParserConfig | ((ast: OfficeParserAST, err?: any) => void), config?: OfficeParserConfig): Promise<OfficeParserAST>;
|
|
1550
1667
|
/**
|
|
1551
1668
|
* Terminates all active OCR workers and cleans up resources.
|
|
1552
1669
|
*
|
|
@@ -1618,7 +1735,7 @@ export declare class OfficeConverter {
|
|
|
1618
1735
|
* });
|
|
1619
1736
|
* ```
|
|
1620
1737
|
*/
|
|
1621
|
-
static convert<F extends string | Buffer | ArrayBuffer, T extends SupportedFileType = InferFileTypeFromPath<F>>(file: F, destination: SupportedDestination<T>, config?: OfficeConverterConfig<SupportedDestination<T>, T>): Promise<ConversionResult<SupportedDestination<T>>>;
|
|
1738
|
+
static convert<F extends string | Buffer | ArrayBuffer | Uint8Array, T extends SupportedFileType = InferFileTypeFromPath<F>>(file: F, destination: SupportedDestination<T>, config?: OfficeConverterConfig<SupportedDestination<T>, T>): Promise<ConversionResult<SupportedDestination<T>>>;
|
|
1622
1739
|
}
|
|
1623
1740
|
export declare const parseOffice: typeof OfficeParser.parseOffice;
|
|
1624
1741
|
export declare const terminateOcr: typeof OfficeParser.terminateOcr;
|