officeparser 7.0.3 → 7.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/types.d.ts CHANGED
@@ -28,7 +28,9 @@ export declare enum OfficeErrorType {
28
28
  /** Output mapping in style mapping is invalid */
29
29
  INVALID_OUTPUT_MAPPING = "INVALID_OUTPUT_MAPPING",
30
30
  /** Semantic chunking strategy is selected but no embedding function is provided */
31
- MISSING_EMBEDDING_FUNCTION = "MISSING_EMBEDDING_FUNCTION"
31
+ MISSING_EMBEDDING_FUNCTION = "MISSING_EMBEDDING_FUNCTION",
32
+ /** The operation was aborted */
33
+ OPERATION_ABORTED = "OPERATION_ABORTED"
32
34
  }
33
35
  /**
34
36
  * Standard warning types for OfficeParser.
@@ -68,6 +70,68 @@ export declare enum OfficeWarningType {
68
70
  /** A node was skipped because it only contained whitespace */
69
71
  WHITESPACE_NODE_SKIPPED = "WHITESPACE_NODE_SKIPPED"
70
72
  }
73
+ /**
74
+ * Consolidated timeout settings for OCR operations.
75
+ * Preferred over the individual flat timeout properties on {@link OcrConfig},
76
+ * which are now deprecated.
77
+ *
78
+ * If a key is present here, it takes priority over the corresponding deprecated
79
+ * flat property (e.g. `timeout.autoTerminate` wins over `autoTerminateTimeout`).
80
+ * Set any value to `0` to disable that specific timeout.
81
+ */
82
+ export interface OcrTimeoutConfig {
83
+ /**
84
+ * Timeout in milliseconds of inactivity before the OCR worker pool is
85
+ * automatically terminated and freed.
86
+ *
87
+ * The timer resets every time a new OCR job is enqueued. When the last
88
+ * job completes and this duration passes without a new one, the entire
89
+ * worker pool is torn down so that no background threads keep the Node.js
90
+ * process alive unnecessarily.
91
+ *
92
+ * Set to `0` to keep workers alive indefinitely (useful when you want to
93
+ * call {@link terminateOcr} manually at shutdown time).
94
+ * Default is 10,000 ms (10 seconds).
95
+ */
96
+ autoTerminate?: number;
97
+ /**
98
+ * Timeout in milliseconds for initializing a Tesseract worker
99
+ * (loading the JS runtime, downloading or loading the `.traineddata`
100
+ * language file) or for re-initializing an existing worker with a
101
+ * different language.
102
+ *
103
+ * Multi-language combinations (e.g. `'por+eng+spa'`) must download a
104
+ * separate `.traineddata` file for each language and are therefore
105
+ * particularly susceptible to slow networks. Tune this value upward if
106
+ * your OCR environment has high network latency or if you are loading
107
+ * languages from disk in a large container image.
108
+ *
109
+ * When the timeout fires, the failed job is rejected with a non-fatal
110
+ * {@link OfficeWarningType.OCR_FAILED} warning and parsing continues
111
+ * without OCR output for that image. The stalled worker is terminated
112
+ * and removed from the pool to prevent thread leaks.
113
+ *
114
+ * Set to `0` to wait indefinitely (not recommended for production; a hung
115
+ * network request will block the entire OCR queue for that language).
116
+ * Default is 60,000 ms (60 seconds).
117
+ */
118
+ workerLoad?: number;
119
+ /**
120
+ * Timeout in milliseconds for the actual OCR text-recognition call
121
+ * (`worker.recognize(image)`) on an already-initialized Tesseract worker.
122
+ *
123
+ * Recognition time scales with image resolution and the number of active
124
+ * languages. Very high-resolution scans or unusual character sets can
125
+ * exceed the default. If this timeout fires, the job is rejected with a
126
+ * non-fatal {@link OfficeWarningType.OCR_FAILED} warning; the worker is
127
+ * terminated and evicted from the pool because its internal state after a
128
+ * mid-recognition timeout is undefined.
129
+ *
130
+ * Set to `0` to wait indefinitely.
131
+ * Default is 30,000 ms (30 seconds).
132
+ */
133
+ recognition?: number;
134
+ }
71
135
  /**
72
136
  * Configuration options for OCR.
73
137
  */
@@ -102,11 +166,33 @@ export interface OcrConfig {
102
166
  */
103
167
  langPath?: string;
104
168
  /**
169
+ * Consolidated timeout settings for all OCR operations.
170
+ *
171
+ * Prefer this over the deprecated flat timeout properties.
172
+ * If `timeout.autoTerminate` is set, it takes priority over the deprecated `autoTerminateTimeout`.
173
+ */
174
+ timeout?: OcrTimeoutConfig;
175
+ /**
176
+ * @deprecated Use `timeout.autoTerminate` instead.
177
+ *
105
178
  * Timeout in milliseconds of inactivity before the OCR worker pool is automatically terminated.
106
179
  * Set to 0 to disable auto-termination.
107
180
  * Default is 10,000 (10 seconds).
181
+ *
182
+ * If `timeout.autoTerminate` is also set, that value takes priority over this one.
108
183
  */
109
184
  autoTerminateTimeout?: number;
185
+ /**
186
+ * An optional AbortSignal propagated from the main parser configuration to abort active OCR jobs.
187
+ * If the signal is aborted:
188
+ * 1. Any pending OCR jobs in the scheduler queue are rejected immediately.
189
+ * 2. Any active OCR job running on a Tesseract worker will reject, the worker will be
190
+ * terminated, and it will be removed from the pool to avoid hanging worker threads.
191
+ *
192
+ * Developers should prefer passing this at the top level of `parseOffice` (as `config.abortSignal`),
193
+ * which automatically propagates here.
194
+ */
195
+ abortSignal?: AbortSignal | null;
110
196
  }
111
197
  /**
112
198
  * Configuration options for the OfficeParser.
@@ -173,6 +259,20 @@ export interface OfficeParserConfig {
173
259
  * If provided, `ocrLanguage` will be ignored in favor of `ocrConfig.language`.
174
260
  */
175
261
  ocrConfig?: OcrConfig;
262
+ /**
263
+ * An optional AbortSignal to cancel the parsing operation.
264
+ * When aborted, the parser immediately rejects with a standard AbortError (DOMException).
265
+ *
266
+ * ### Format-Specific Abort Behavior:
267
+ * - **PDF**: Checked between page loads and before individual image OCR operations.
268
+ * - **RTF**: Checked before parsing/traversal and before running OCR on image attachments.
269
+ * - **DOCX/XLSX/PPTX/ODF**: Checked during zip decompression before loading and parsing XML files.
270
+ * - **CSV/MD/HTML**: Checked at the start of the parsing phase.
271
+ *
272
+ * Note: If an OCR operation is currently running on a Tesseract worker when aborted,
273
+ * the worker will be terminated and removed from the worker pool automatically to prevent leaks.
274
+ */
275
+ abortSignal?: AbortSignal | null;
176
276
  /**
177
277
  * Flag to serialize raw content (XML) as clean, formatted strings.
178
278
  * Only relevant when `includeRawContent` is true.
@@ -366,6 +466,12 @@ export interface CommonGeneratorConfig {
366
466
  * Defaults to false.
367
467
  */
368
468
  ignoreInternalLinks?: boolean;
469
+ /**
470
+ * An optional AbortSignal to cancel the generation operation.
471
+ * When aborted, the generator immediately rejects with a standard AbortError.
472
+ * Currently supported by PdfGenerator and ChunkingGenerator.
473
+ */
474
+ abortSignal?: AbortSignal | null;
369
475
  }
370
476
  /**
371
477
  * Destination-aware generator configuration.
@@ -494,6 +600,12 @@ export interface PdfGeneratorConfig {
494
600
  * Useful for setting custom executable paths or args in CI/CD.
495
601
  */
496
602
  launchOptions?: any;
603
+ /**
604
+ * Timeout in milliseconds for PDF generation.
605
+ * Limits the time spent waiting for Puppeteer to launch, load content, and render PDF.
606
+ * Defaults to 30000 ms (30 seconds). Set to 0 to disable.
607
+ */
608
+ timeout?: number;
497
609
  }
498
610
  /**
499
611
  * Structured style mapping definition for the StyleMapper.
@@ -799,6 +911,11 @@ export interface SemanticChunkingConfig extends BaseChunkingConfig {
799
911
  * Default is 50.
800
912
  */
801
913
  embeddingBatchSize?: number;
914
+ /**
915
+ * Timeout in milliseconds for individual embedding API calls.
916
+ * Defaults to 10000 ms (10 seconds). Set to 0 to disable.
917
+ */
918
+ timeout?: number;
802
919
  }
803
920
  /**
804
921
  * Discriminated union of all chunking strategy configurations.
package/dist/types.js CHANGED
@@ -33,6 +33,8 @@ var OfficeErrorType;
33
33
  OfficeErrorType["INVALID_OUTPUT_MAPPING"] = "INVALID_OUTPUT_MAPPING";
34
34
  /** Semantic chunking strategy is selected but no embedding function is provided */
35
35
  OfficeErrorType["MISSING_EMBEDDING_FUNCTION"] = "MISSING_EMBEDDING_FUNCTION";
36
+ /** The operation was aborted */
37
+ OfficeErrorType["OPERATION_ABORTED"] = "OPERATION_ABORTED";
36
38
  })(OfficeErrorType || (exports.OfficeErrorType = OfficeErrorType = {}));
37
39
  /**
38
40
  * Standard warning types for OfficeParser.
@@ -66,12 +66,25 @@ function resolveParserConfig(userConfig) {
66
66
  const { ocrConfig, ...rest } = userConfig;
67
67
  Object.assign(config, rest);
68
68
  if (ocrConfig) {
69
- config.ocrConfig = { ...config.ocrConfig, ...ocrConfig };
69
+ const { timeout, ...ocrRest } = ocrConfig;
70
+ config.ocrConfig = {
71
+ ...config.ocrConfig,
72
+ ...ocrRest,
73
+ timeout: {
74
+ autoTerminate: timeout?.autoTerminate !== undefined ? timeout.autoTerminate : config.ocrConfig.timeout.autoTerminate,
75
+ workerLoad: timeout?.workerLoad !== undefined ? timeout.workerLoad : config.ocrConfig.timeout.workerLoad,
76
+ recognition: timeout?.recognition !== undefined ? timeout.recognition : config.ocrConfig.timeout.recognition,
77
+ }
78
+ };
70
79
  }
71
80
  // 3. Handle legacy ocrLanguage mapping if not explicitly set in ocrConfig
72
81
  if (userConfig.ocrLanguage && !userConfig.ocrConfig?.language) {
73
82
  config.ocrConfig.language = userConfig.ocrLanguage;
74
83
  }
84
+ // 4. Propagate the top-level abortSignal to ocrConfig so the OCR subsystem is aware of it
85
+ if (config.abortSignal) {
86
+ config.ocrConfig.abortSignal = config.abortSignal;
87
+ }
75
88
  return config;
76
89
  }
77
90
  /**
@@ -27,6 +27,11 @@ export declare const getOfficeError: (type: OfficeErrorType, config?: OfficePars
27
27
  * Wraps an existing error with OfficeParser context and performs corruption detection.
28
28
  * Optionally logs the error to console.
29
29
  *
30
+ * **Important**: Do NOT pass AbortErrors to this function. AbortErrors (err.name === 'AbortError')
31
+ * represent deliberate user cancellation and must be re-thrown as-is from the catch block so that
32
+ * callers can reliably detect them via `err.name === 'AbortError'` or `err instanceof DOMException`.
33
+ * This function always returns a plain `new Error(...)`, which would strip the AbortError identity.
34
+ *
30
35
  * @param error - The original error object
31
36
  * @param config - Parser configuration
32
37
  * @param filePath - Optional file path for context
@@ -44,3 +49,18 @@ export declare const getWrappedError: (error: any, config: OfficeParserConfig, f
44
49
  * @param error - Optional original error object
45
50
  */
46
51
  export declare const logWarning: (type: OfficeWarningType, config?: OfficeParserConfig, info?: any, error?: any) => void;
52
+ /**
53
+ * Creates and returns a standard AbortError (DOMException if available).
54
+ * Used when the user signals cancellation of the parser operation.
55
+ *
56
+ * @returns Error object representing the abort action
57
+ */
58
+ export declare const getAbortError: () => Error;
59
+ /**
60
+ * Checks the provided AbortSignal and throws an AbortError if it was aborted.
61
+ * Helps cleanly interrupt loops and asynchronous phases of parsing.
62
+ *
63
+ * @param signal - Optional AbortSignal to inspect
64
+ * @throws {DOMException} If the signal has been aborted
65
+ */
66
+ export declare const checkAbortSignal: (signal?: AbortSignal | null) => void;
@@ -7,7 +7,7 @@
7
7
  * consistent error reporting across all parsers and the main entry point.
8
8
  */
9
9
  Object.defineProperty(exports, "__esModule", { value: true });
10
- exports.logWarning = exports.getWrappedError = exports.getOfficeError = exports.getWarningMessage = void 0;
10
+ exports.checkAbortSignal = exports.getAbortError = exports.logWarning = exports.getWrappedError = exports.getOfficeError = exports.getWarningMessage = void 0;
11
11
  const types_js_1 = require("../types.js");
12
12
  /** Error header prefix for all error messages */
13
13
  const ERRORHEADER = "[OfficeParser]: ";
@@ -28,7 +28,8 @@ const ERROR_MESSAGES = {
28
28
  [types_js_1.OfficeErrorType.INVALID_STYLE_MAPPING]: (mapping) => `Invalid style mapping string: ${mapping}`,
29
29
  [types_js_1.OfficeErrorType.INVALID_SELECTOR]: (selector) => `Invalid selector: ${selector}`,
30
30
  [types_js_1.OfficeErrorType.INVALID_OUTPUT_MAPPING]: (output) => `Invalid output mapping: ${output}`,
31
- [types_js_1.OfficeErrorType.MISSING_EMBEDDING_FUNCTION]: `Semantic chunking requires an "embeddingFunction" to be provided in chunksConfig. This function must accept a string and return a Promise resolving to a number array (vector).`
31
+ [types_js_1.OfficeErrorType.MISSING_EMBEDDING_FUNCTION]: `Semantic chunking requires an "embeddingFunction" to be provided in chunksConfig. This function must accept a string and return a Promise resolving to a number array (vector).`,
32
+ [types_js_1.OfficeErrorType.OPERATION_ABORTED]: `The operation was aborted.`
32
33
  };
33
34
  /**
34
35
  * Lookup table for warning messages.
@@ -118,6 +119,11 @@ exports.getOfficeError = getOfficeError;
118
119
  * Wraps an existing error with OfficeParser context and performs corruption detection.
119
120
  * Optionally logs the error to console.
120
121
  *
122
+ * **Important**: Do NOT pass AbortErrors to this function. AbortErrors (err.name === 'AbortError')
123
+ * represent deliberate user cancellation and must be re-thrown as-is from the catch block so that
124
+ * callers can reliably detect them via `err.name === 'AbortError'` or `err instanceof DOMException`.
125
+ * This function always returns a plain `new Error(...)`, which would strip the AbortError identity.
126
+ *
121
127
  * @param error - The original error object
122
128
  * @param config - Parser configuration
123
129
  * @param filePath - Optional file path for context
@@ -176,3 +182,32 @@ const logWarning = (type, config, info, error) => {
176
182
  reportIssue(issue, config);
177
183
  };
178
184
  exports.logWarning = logWarning;
185
+ /**
186
+ * Creates and returns a standard AbortError (DOMException if available).
187
+ * Used when the user signals cancellation of the parser operation.
188
+ *
189
+ * @returns Error object representing the abort action
190
+ */
191
+ const getAbortError = () => {
192
+ const message = ERROR_MESSAGES[types_js_1.OfficeErrorType.OPERATION_ABORTED];
193
+ if (typeof DOMException !== 'undefined') {
194
+ return new DOMException(message, 'AbortError');
195
+ }
196
+ const err = new Error(message);
197
+ err.name = 'AbortError';
198
+ return err;
199
+ };
200
+ exports.getAbortError = getAbortError;
201
+ /**
202
+ * Checks the provided AbortSignal and throws an AbortError if it was aborted.
203
+ * Helps cleanly interrupt loops and asynchronous phases of parsing.
204
+ *
205
+ * @param signal - Optional AbortSignal to inspect
206
+ * @throws {DOMException} If the signal has been aborted
207
+ */
208
+ const checkAbortSignal = (signal) => {
209
+ if (signal?.aborted) {
210
+ throw (0, exports.getAbortError)();
211
+ }
212
+ };
213
+ exports.checkAbortSignal = checkAbortSignal;
@@ -14,7 +14,7 @@ exports.loadPdfJs = loadPdfJs;
14
14
  const envUtils_js_1 = require("./envUtils.js");
15
15
  async function loadNodeEsmModule(specifier) {
16
16
  // In Node.js, we resolve the specifier to an absolute file URL.
17
- // This ensures that the dynamic import() call (executed via new Function)
17
+ // This ensures that the dynamic import() call
18
18
  // always finds the correct module regardless of the caller's context.
19
19
  // This is especially important in Node 18 for sub-paths of packages.
20
20
  try {
@@ -22,11 +22,11 @@ async function loadNodeEsmModule(specifier) {
22
22
  // @ts-ignore - require.resolve is available in Node.js
23
23
  const absolutePath = require.resolve(specifier);
24
24
  const fileUrl = pathToFileURL(absolutePath).href;
25
- return new Function('s', 'return import(s)')(fileUrl);
25
+ return import(fileUrl);
26
26
  }
27
27
  catch (e) {
28
28
  // Fallback for cases where require.resolve might fail (e.g. non-file specifiers)
29
- return new Function('s', 'return import(s)')(specifier);
29
+ return import(specifier);
30
30
  }
31
31
  }
32
32
  /**