officeparser 6.0.7 → 6.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/README.md +136 -52
  2. package/dist/OfficeParser.d.ts +10 -1
  3. package/dist/OfficeParser.js +44 -56
  4. package/dist/cli.d.ts +20 -0
  5. package/dist/cli.js +117 -0
  6. package/dist/index.d.ts +4 -4
  7. package/dist/index.js +7 -59
  8. package/dist/index.mjs +18 -0
  9. package/dist/officeparser.browser.d.ts +133 -3
  10. package/dist/officeparser.browser.iife.js +115 -0
  11. package/dist/officeparser.browser.mjs +114 -0
  12. package/dist/parsers/ExcelParser.d.ts +1 -1
  13. package/dist/parsers/ExcelParser.js +76 -68
  14. package/dist/parsers/OpenOfficeParser.d.ts +1 -1
  15. package/dist/parsers/OpenOfficeParser.js +224 -159
  16. package/dist/parsers/PdfParser.d.ts +1 -1
  17. package/dist/parsers/PdfParser.js +98 -94
  18. package/dist/parsers/PowerPointParser.d.ts +1 -1
  19. package/dist/parsers/PowerPointParser.js +188 -179
  20. package/dist/parsers/RtfParser.d.ts +21 -1
  21. package/dist/parsers/RtfParser.js +117 -48
  22. package/dist/parsers/WordParser.d.ts +2 -1
  23. package/dist/parsers/WordParser.js +214 -123
  24. package/dist/sbom.cdx.json +1807 -0
  25. package/dist/types.d.ts +123 -3
  26. package/dist/utils/chartUtils.js +2 -0
  27. package/dist/utils/dateUtils.d.ts +17 -0
  28. package/dist/utils/dateUtils.js +69 -0
  29. package/dist/utils/envUtils.d.ts +24 -0
  30. package/dist/utils/envUtils.js +69 -0
  31. package/dist/utils/moduleLoader.d.ts +2 -1
  32. package/dist/utils/moduleLoader.js +9 -39
  33. package/dist/utils/ocrUtils.d.ts +16 -12
  34. package/dist/utils/ocrUtils.js +186 -25
  35. package/dist/utils/xmlUtils.d.ts +80 -9
  36. package/dist/utils/xmlUtils.js +236 -18
  37. package/dist/utils/zipUtils.js +6 -47
  38. package/package.json +31 -16
  39. package/dist/officeParserBundle@6.0.7.js +0 -154
  40. package/dist/officeparser.browser.js +0 -154
package/dist/types.d.ts CHANGED
@@ -1,3 +1,40 @@
1
+ /**
2
+ * Configuration options for OCR.
3
+ */
4
+ export interface OcrConfig {
5
+ /**
6
+ * Language for OCR.
7
+ * Default is 'eng'.
8
+ *
9
+ * You can provide multiple languages separated by a `+` sign (e.g., 'eng+fra' for English and French).
10
+ * The OCR engine will then attempt to recognize text in any of the specified languages.
11
+ *
12
+ * See the list of supported languages and their codes here:
13
+ * https://tesseract-ocr.github.io/tessdoc/Data-Files#data-files-for-version-400-november-29-2016
14
+ */
15
+ language?: string;
16
+ /**
17
+ * Path to the Tesseract worker script.
18
+ * Primarily used for offline/air-gapped environments.
19
+ */
20
+ workerPath?: string;
21
+ /**
22
+ * Path to the Tesseract core script.
23
+ * Primarily used for offline/air-gapped environments.
24
+ */
25
+ corePath?: string;
26
+ /**
27
+ * Path for Tesseract language files (traineddata).
28
+ * Primarily used for offline/air-gapped environments.
29
+ */
30
+ langPath?: string;
31
+ /**
32
+ * Timeout in milliseconds of inactivity before the OCR worker pool is automatically terminated.
33
+ * Set to 0 to disable auto-termination.
34
+ * Default is 10,000 (10 seconds).
35
+ */
36
+ autoTerminateTimeout?: number;
37
+ }
1
38
  /**
2
39
  * Configuration options for the OfficeParser.
3
40
  */
@@ -40,6 +77,7 @@ export interface OfficeParserConfig {
40
77
  */
41
78
  ocr?: boolean;
42
79
  /**
80
+ * @deprecated Use `ocrConfig.language` instead.
43
81
  * Language for OCR.
44
82
  * Default is 'eng'.
45
83
  *
@@ -50,14 +88,41 @@ export interface OfficeParserConfig {
50
88
  * https://tesseract-ocr.github.io/tessdoc/Data-Files#data-files-for-version-400-november-29-2016
51
89
  */
52
90
  ocrLanguage?: string;
91
+ /**
92
+ * Shared OCR configuration for worker pooling and offline support.
93
+ * If provided, `ocrLanguage` will be ignored in favor of `ocrConfig.language`.
94
+ */
95
+ ocrConfig?: OcrConfig;
96
+ /**
97
+ * Flag to serialize raw content (XML) as clean, formatted strings.
98
+ * Only relevant when `includeRawContent` is true.
99
+ * Default is true.
100
+ *
101
+ * If false, the parser will attempt to extract the original raw substring from the
102
+ * source document instead of re-serializing the DOM node.
103
+ */
104
+ serializeRawContent?: boolean;
105
+ /**
106
+ * Flag to preserve original XML whitespace and line endings when serializing.
107
+ * Only relevant when `includeRawContent` is true and `serializeRawContent` is true.
108
+ * Default is false.
109
+ */
110
+ preserveXmlWhitespace?: boolean;
53
111
  /**
54
112
  * The URL/path to the PDF.js worker script.
55
113
  *
56
114
  * **Mandatory** when using PDF parsing in browser environments to avoid worker configuration errors.
57
- * If not provided, it defaults to `https://unpkg.com/pdfjs-dist@5.5.207/build/pdf.worker.min.mjs`.
115
+ * If not provided, it defaults to `https://unpkg.com/pdfjs-dist@5.6.205/build/pdf.worker.min.mjs`.
58
116
  * You can override this with your own local path or a different CDN link.
59
117
  */
60
118
  pdfWorkerSrc?: string;
119
+ /**
120
+ * Flag to include break nodes in the AST.
121
+ * This is currently only supported for Word documents. (w:br nodes)
122
+ *
123
+ * Default is false
124
+ */
125
+ includeBreakNodes?: boolean;
61
126
  }
62
127
  /**
63
128
  * Supported file types for parsing.
@@ -66,7 +131,7 @@ export type SupportedFileType = 'docx' | 'pptx' | 'xlsx' | 'odt' | 'odp' | 'ods'
66
131
  /**
67
132
  * Types of content nodes in the AST.
68
133
  */
69
- export type OfficeContentNodeType = 'paragraph' | 'heading' | 'table' | 'list' | 'text' | 'image' | 'chart' | 'drawing' | 'slide' | 'note' | 'sheet' | 'row' | 'cell' | 'page';
134
+ export type OfficeContentNodeType = 'paragraph' | 'heading' | 'table' | 'list' | 'text' | 'image' | 'chart' | 'drawing' | 'slide' | 'note' | 'sheet' | 'row' | 'cell' | 'page' | 'break';
70
135
  /**
71
136
  * Supported MIME types for attachments.
72
137
  */
@@ -169,6 +234,20 @@ export interface SheetMetadata {
169
234
  /** The style of the sheet. */
170
235
  style?: string;
171
236
  }
237
+ /**
238
+ * Detailed indentation information for paragraphs and headings.
239
+ * Values are typically in twentieths of a point (twips) in OOXML.
240
+ */
241
+ export interface IndentationMetadata {
242
+ /** Left indentation. */
243
+ left?: number;
244
+ /** Right indentation. */
245
+ right?: number;
246
+ /** First line indentation. */
247
+ firstLine?: number;
248
+ /** Hanging indentation. */
249
+ hanging?: number;
250
+ }
172
251
  /**
173
252
  * Metadata for a heading.
174
253
  */
@@ -179,6 +258,8 @@ export interface HeadingMetadata {
179
258
  alignment?: 'left' | 'center' | 'right' | 'justify';
180
259
  /** The style of the heading. */
181
260
  style?: string;
261
+ /** Detailed indentation information. */
262
+ paragraphIndentation?: IndentationMetadata;
182
263
  }
183
264
  /**
184
265
  * Metadata for a paragraph.
@@ -188,6 +269,8 @@ export interface ParagraphMetadata {
188
269
  alignment?: 'left' | 'center' | 'right' | 'justify';
189
270
  /** The style of the paragraph. */
190
271
  style?: string;
272
+ /** Detailed indentation information. */
273
+ paragraphIndentation?: IndentationMetadata;
191
274
  }
192
275
  /**
193
276
  * Metadata for a list item.
@@ -203,6 +286,8 @@ export interface ListMetadata {
203
286
  * @example 0 for top-level items, 1 for first nested level
204
287
  */
205
288
  indentation: number;
289
+ /** Detailed indentation information. */
290
+ paragraphIndentation?: IndentationMetadata;
206
291
  /**
207
292
  * Text alignment of the list item.
208
293
  * @example 'left', 'center', 'right', 'justify'
@@ -329,10 +414,35 @@ export interface NoteMetadata {
329
414
  */
330
415
  noteId?: string;
331
416
  }
417
+ /**
418
+ * Metadata for break nodes.
419
+ * Used in DOCX files to track line and page breaks.
420
+ */
421
+ export interface BreakMetadata {
422
+ /**
423
+ * Type of break. The break type determines the next location where
424
+ * text shall be placed.
425
+ * - 'column': The next text will be placed in the next column.
426
+ * - 'page': The next text will be placed on the next page.
427
+ * - 'lastRenderedPage': The editing application has inserted a soft break on the last save.
428
+ * - 'textWrapping' (default, assumed when not specified): The next text will be placed on the next line.
429
+ * - 'carriageReturn': An explicit carriage return (w:cr) equivalent to a hard line break.
430
+ */
431
+ breakType: 'column' | 'page' | 'lastRenderedPage' | 'textWrapping' | 'carriageReturn';
432
+ /**
433
+ * Specifies the location which shall be used as the next available line when breakType
434
+ * has a value of 'textWrapping'. Should be ignored for other break types.
435
+ * - 'all': text wrapping break shall advance the text to the next line which spans the full width of the line
436
+ * - 'left': text wrapping break shall restart in next text region unblocked on the left
437
+ * - 'none': text wrapping break shall advance the text to the next line regardless of any floating objects
438
+ * - 'right': text wrapping break shall restart in next text region unblocked on the right
439
+ */
440
+ clear?: 'all' | 'left' | 'none' | 'right';
441
+ }
332
442
  /**
333
443
  * Union type for content metadata.
334
444
  */
335
- export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata | ListMetadata | CellMetadata | ImageMetadata | ChartMetadata | PageMetadata | ParagraphMetadata | TextMetadata | NoteMetadata | undefined;
445
+ export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata | ListMetadata | CellMetadata | ImageMetadata | ChartMetadata | PageMetadata | ParagraphMetadata | TextMetadata | NoteMetadata | BreakMetadata | undefined;
336
446
  /**
337
447
  * Represents a node in the document content tree.
338
448
  * This is the core building block of the parsed document structure.
@@ -540,6 +650,16 @@ export interface OfficeMetadata {
540
650
  formatting?: Partial<TextFormatting>;
541
651
  /** Style map for styles in the document. */
542
652
  styleMap?: Record<string, Partial<TextFormatting>>;
653
+ /**
654
+ * User-defined custom properties embedded in the document.
655
+ * Sources by format:
656
+ * - DOCX/XLSX/PPTX: `docProps/custom.xml` (Office custom document properties)
657
+ * - ODT/ODP/ODS: `meta:user-defined` elements in `meta.xml`
658
+ * - PDF: non-standard entries in the PDF Info dictionary
659
+ * RTF does not support custom properties; the `\info` group is not extracted.
660
+ * Values are typed as string, number, boolean, or Date where the source format provides type information.
661
+ */
662
+ customProperties?: Record<string, string | number | boolean | Date>;
543
663
  }
544
664
  /**
545
665
  * The Abstract Syntax Tree (AST) returned by the parser.
@@ -43,6 +43,8 @@ const extractOpenXmlChartData = (xmlBuffer) => {
43
43
  const xml = xmlBuffer.toString("utf8");
44
44
  const dom = (0, xmlUtils_1.parseXmlString)(xml);
45
45
  const root = dom.documentElement;
46
+ if (!root)
47
+ return { title: undefined, xAxisTitle: undefined, yAxisTitle: undefined, dataSets: [], labels: [], rawTexts: [] };
46
48
  const title = extractOpenXmlRichText(root, "c:title");
47
49
  // Axis titles
48
50
  let xAxisTitle = undefined;
@@ -0,0 +1,17 @@
1
+ /**
2
+ * Date Parsing Utilities
3
+ *
4
+ * Provides robust functions for parsing date strings from various office formats.
5
+ * Handles standard ISO dates, PDF-specific date formats, and malformed strings.
6
+ *
7
+ * @module dateUtils
8
+ */
9
+ /**
10
+ * Parses a date string into a Date object.
11
+ * Handles standard ISO formats and falls back to native parsing.
12
+ * Returns undefined instead of "Invalid Date" if parsing fails.
13
+ *
14
+ * @param dateString - The date string to parse
15
+ * @returns Parsed Date object or undefined if parsing fails
16
+ */
17
+ export declare function parseOfficeDate(dateString: string | undefined): Date | undefined;
@@ -0,0 +1,69 @@
1
+ "use strict";
2
+ /**
3
+ * Date Parsing Utilities
4
+ *
5
+ * Provides robust functions for parsing date strings from various office formats.
6
+ * Handles standard ISO dates, PDF-specific date formats, and malformed strings.
7
+ *
8
+ * @module dateUtils
9
+ */
10
+ Object.defineProperty(exports, "__esModule", { value: true });
11
+ exports.parseOfficeDate = parseOfficeDate;
12
+ /**
13
+ * Parses a date string into a Date object.
14
+ * Handles standard ISO formats and falls back to native parsing.
15
+ * Returns undefined instead of "Invalid Date" if parsing fails.
16
+ *
17
+ * @param dateString - The date string to parse
18
+ * @returns Parsed Date object or undefined if parsing fails
19
+ */
20
+ function parseOfficeDate(dateString) {
21
+ if (!dateString)
22
+ return undefined;
23
+ try {
24
+ // PDF-specific format detection: D:YYYYMMDDHHmmSSOHH'mm'
25
+ if (dateString.startsWith('D:')) {
26
+ return parsePdfDate(dateString);
27
+ }
28
+ const date = new Date(dateString);
29
+ return isNaN(date.getTime()) ? undefined : date;
30
+ }
31
+ catch {
32
+ return undefined;
33
+ }
34
+ }
35
+ /**
36
+ * Internal helper for PDF-specific date format: D:YYYYMMDDHHmmSSOHH'mm'
37
+ * @param dateString - The PDF date string
38
+ */
39
+ function parsePdfDate(dateString) {
40
+ try {
41
+ // Remove "D:" prefix
42
+ let str = dateString.slice(2);
43
+ // Extract components: YYYYMMDDHHmmSS
44
+ const year = parseInt(str.slice(0, 4), 10);
45
+ const month = parseInt(str.slice(4, 6), 10) - 1; // 0-indexed
46
+ const day = parseInt(str.slice(6, 8), 10) || 1;
47
+ const hour = parseInt(str.slice(8, 10), 10) || 0;
48
+ const minute = parseInt(str.slice(10, 12), 10) || 0;
49
+ const second = parseInt(str.slice(12, 14), 10) || 0;
50
+ // Handle timezone if present
51
+ const tzMatch = str.slice(14).match(/([+-Z])(\d{2})'?(\d{2})?'?/);
52
+ if (tzMatch) {
53
+ if (tzMatch[1] === 'Z') {
54
+ return new Date(Date.UTC(year, month, day, hour, minute, second));
55
+ }
56
+ const tzSign = tzMatch[1] === '-' ? -1 : 1;
57
+ const tzHours = parseInt(tzMatch[2], 10) || 0;
58
+ const tzMinutes = parseInt(tzMatch[3], 10) || 0;
59
+ const offset = tzSign * (tzHours * 60 + tzMinutes);
60
+ // Create date in UTC and adjust for timezone
61
+ const utc = Date.UTC(year, month, day, hour, minute, second);
62
+ return new Date(utc - offset * 60000);
63
+ }
64
+ return new Date(year, month, day, hour, minute, second);
65
+ }
66
+ catch {
67
+ return undefined;
68
+ }
69
+ }
@@ -0,0 +1,24 @@
1
+ /**
2
+ * Environment detection and safe utility wrappers.
3
+ */
4
+ /**
5
+ * Detect if we are running in a browser environment.
6
+ */
7
+ export declare const isBrowser: boolean;
8
+ /**
9
+ * Supported Node.js-only features that require explicit guarding for browser compatibility.
10
+ */
11
+ export type NodeFeature = 'fs' | 'path-parsing' | 'pdf-worker-auto-resolution';
12
+ /**
13
+ * Throws an error if attempted to use Node.js-specific features in the browser.
14
+ *
15
+ * @param feature - The Node.js feature being accessed
16
+ * @throws {Error} Clear error message directing browser users to use Buffers
17
+ */
18
+ export declare function assertNode(feature: NodeFeature): void;
19
+ /**
20
+ * Polyfills DOMMatrix if not available globally (required for Node.js < 20).
21
+ * This shim provides enough properties for pdfjs-dist 5.x to calculate
22
+ * text coordinates and transformations.
23
+ */
24
+ export declare function ensureDomMatrix(): void;
@@ -0,0 +1,69 @@
1
+ "use strict";
2
+ /**
3
+ * Environment detection and safe utility wrappers.
4
+ */
5
+ Object.defineProperty(exports, "__esModule", { value: true });
6
+ exports.isBrowser = void 0;
7
+ exports.assertNode = assertNode;
8
+ exports.ensureDomMatrix = ensureDomMatrix;
9
+ /**
10
+ * Detect if we are running in a browser environment.
11
+ */
12
+ exports.isBrowser = typeof window !== 'undefined' && typeof window.document !== 'undefined';
13
+ /**
14
+ * Human-readable descriptions for Node-only features.
15
+ */
16
+ const readableFeatures = {
17
+ 'fs': 'direct file system access',
18
+ 'path-parsing': 'parsing from file path string',
19
+ 'pdf-worker-auto-resolution': 'automatic PDF worker resolution from node_modules'
20
+ };
21
+ /**
22
+ * Throws an error if attempted to use Node.js-specific features in the browser.
23
+ *
24
+ * @param feature - The Node.js feature being accessed
25
+ * @throws {Error} Clear error message directing browser users to use Buffers
26
+ */
27
+ function assertNode(feature) {
28
+ if (exports.isBrowser) {
29
+ throw new Error(`officeparser: '${readableFeatures[feature]}' is not supported in the browser. Browser users must pass file content as Buffer or ArrayBuffer directly.`);
30
+ }
31
+ }
32
+ /**
33
+ * Polyfills DOMMatrix if not available globally (required for Node.js < 20).
34
+ * This shim provides enough properties for pdfjs-dist 5.x to calculate
35
+ * text coordinates and transformations.
36
+ */
37
+ function ensureDomMatrix() {
38
+ if (typeof global !== 'undefined' && !global.DOMMatrix) {
39
+ global.DOMMatrix = class DOMMatrix {
40
+ a;
41
+ b;
42
+ c;
43
+ d;
44
+ e;
45
+ f;
46
+ constructor(init) {
47
+ if (Array.isArray(init) && init.length >= 6) {
48
+ this.a = init[0];
49
+ this.b = init[1];
50
+ this.c = init[2];
51
+ this.d = init[3];
52
+ this.e = init[4];
53
+ this.f = init[5];
54
+ }
55
+ else {
56
+ this.a = this.d = 1;
57
+ this.b = this.c = this.e = this.f = 0;
58
+ }
59
+ }
60
+ // Map standard matrix properties for compatibility
61
+ get m11() { return this.a; }
62
+ get m12() { return this.b; }
63
+ get m21() { return this.c; }
64
+ get m22() { return this.d; }
65
+ get m41() { return this.e; }
66
+ get m42() { return this.f; }
67
+ };
68
+ }
69
+ }
@@ -7,10 +7,11 @@
7
7
  * This approach prevents TypeScript from transpiling dynamic import()
8
8
  * into require() when targeting CommonJS.
9
9
  */
10
+ import type * as FileTypeModule from 'file-type' with { 'resolution-mode': 'import' };
10
11
  /**
11
12
  * Specialized loader for file-type
12
13
  */
13
- export declare function loadFileType(): Promise<typeof import('file-type')>;
14
+ export declare function loadFileType(): Promise<typeof FileTypeModule>;
14
15
  /**
15
16
  * Specialized loader for pdfjs-dist
16
17
  */
@@ -8,42 +8,10 @@
8
8
  * This approach prevents TypeScript from transpiling dynamic import()
9
9
  * into require() when targeting CommonJS.
10
10
  */
11
- var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
12
- if (k2 === undefined) k2 = k;
13
- var desc = Object.getOwnPropertyDescriptor(m, k);
14
- if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
15
- desc = { enumerable: true, get: function() { return m[k]; } };
16
- }
17
- Object.defineProperty(o, k2, desc);
18
- }) : (function(o, m, k, k2) {
19
- if (k2 === undefined) k2 = k;
20
- o[k2] = m[k];
21
- }));
22
- var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
23
- Object.defineProperty(o, "default", { enumerable: true, value: v });
24
- }) : function(o, v) {
25
- o["default"] = v;
26
- });
27
- var __importStar = (this && this.__importStar) || (function () {
28
- var ownKeys = function(o) {
29
- ownKeys = Object.getOwnPropertyNames || function (o) {
30
- var ar = [];
31
- for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
32
- return ar;
33
- };
34
- return ownKeys(o);
35
- };
36
- return function (mod) {
37
- if (mod && mod.__esModule) return mod;
38
- var result = {};
39
- if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
40
- __setModuleDefault(result, mod);
41
- return result;
42
- };
43
- })();
44
11
  Object.defineProperty(exports, "__esModule", { value: true });
45
12
  exports.loadFileType = loadFileType;
46
13
  exports.loadPdfJs = loadPdfJs;
14
+ const envUtils_js_1 = require("./envUtils.js");
47
15
  /**
48
16
  * Dynamically loads an ESM module in a Node.js CJS context.
49
17
  *
@@ -59,18 +27,20 @@ async function loadNodeEsmModule(specifier) {
59
27
  * Specialized loader for file-type
60
28
  */
61
29
  async function loadFileType() {
62
- if (typeof window === 'undefined') {
63
- // Node.js path
30
+ if (!envUtils_js_1.isBrowser) {
31
+ // Node.js path: Use dynamic import wrapper for CJS compatibility
64
32
  return loadNodeEsmModule('file-type');
65
33
  }
66
- // Browser path: esbuild handles standard dynamic import()
67
- return Promise.resolve().then(() => __importStar(require('file-type')));
34
+ // Browser path: standard dynamic import() is handled by bundlers (e.g. esbuild/Vite)
35
+ return import('file-type');
68
36
  }
69
37
  /**
70
38
  * Specialized loader for pdfjs-dist
71
39
  */
72
40
  async function loadPdfJs() {
73
- if (typeof window === 'undefined') {
41
+ if (!envUtils_js_1.isBrowser) {
42
+ // Ensure DOMMatrix polyfill for Node.js 18 support
43
+ (0, envUtils_js_1.ensureDomMatrix)();
74
44
  // Node.js environment: require legacy build for stability with ESM-only main
75
45
  try {
76
46
  return await loadNodeEsmModule('pdfjs-dist/legacy/build/pdf.mjs');
@@ -80,5 +50,5 @@ async function loadPdfJs() {
80
50
  }
81
51
  }
82
52
  // Browser environment: esbuild handles standard static-looking dynamic import()
83
- return Promise.resolve().then(() => __importStar(require('pdfjs-dist')));
53
+ return import('pdfjs-dist');
84
54
  }
@@ -4,8 +4,12 @@
4
4
  * This module provides functions for extracting text from images using Tesseract.js.
5
5
  * Used when `config.ocr` is enabled to extract text from embedded images in documents.
6
6
  *
7
+ * Includes a worker pool via OcrSchedulerManager to improve performance when
8
+ * processing multiple images.
9
+ *
7
10
  * @module ocrUtils
8
11
  */
12
+ import { OcrConfig } from '../types.js';
9
13
  /**
10
14
  * Performs Optical Character Recognition (OCR) on an image to extract text.
11
15
  *
@@ -13,26 +17,26 @@
13
17
  * This is useful for extracting text from screenshots, scanned documents,
14
18
  * charts with labels, or any image containing text.
15
19
  *
16
- * The function creates a new Tesseract worker, processes the image,
17
- * and properly terminates the worker to free resources.
20
+ * This function uses a shared worker pool to minimize initialization overhead.
18
21
  *
19
- * @param imageBuffer - The image data as a Node.js Buffer (PNG, JPEG, etc.)
20
- * @param language - The language code for OCR (default: 'eng' for English).
21
- * Supports ISO 639-2/T three-letter codes: 'eng', 'spa', 'fra', 'deu', etc.
22
- * Multiple languages can be combined with '+': 'eng+fra'
22
+ * @param image - The image data as a Buffer, file path, or Blob
23
+ * @param config - Optional configuration for language and custom worker paths
23
24
  * @returns A promise that resolves to the recognized text as a string
24
25
  * @throws {Error} If the image cannot be processed or Tesseract initialization fails
25
26
  *
26
27
  * @example
27
28
  * ```typescript
28
29
  * // Extract text from an English image
29
- * const text = await performOcr(imageBuffer, 'eng');
30
- * console.log(text); // "Annual Revenue: $1.2M"
31
- *
32
- * // Extract text from a multilingual image
33
- * const text = await performOcr(imageBuffer, 'eng+spa');
30
+ * const text = await performOcr(imageBuffer, { language: 'eng' });
34
31
  * ```
35
32
  *
36
33
  * @see https://github.com/naptha/tesseract.js for supported languages and options
37
34
  */
38
- export declare const performOcr: (image: Buffer | string, language?: string) => Promise<string>;
35
+ export declare const performOcr: (image: Buffer | string, config?: OcrConfig) => Promise<string>;
36
+ /**
37
+ * Terminates all OCR workers and cleans up resources.
38
+ *
39
+ * Should be called when the application is shutting down or OCR is no longer needed
40
+ * to prevent memory leaks and dangling worker processes.
41
+ */
42
+ export declare const terminateOcr: () => Promise<void>;