officeparser 6.0.7 → 6.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/README.md +92 -13
  2. package/dist/OfficeParser.d.ts +10 -1
  3. package/dist/OfficeParser.js +43 -56
  4. package/dist/cli.d.ts +20 -0
  5. package/dist/cli.js +116 -0
  6. package/dist/index.d.ts +3 -3
  7. package/dist/index.js +7 -59
  8. package/dist/index.mjs +18 -0
  9. package/dist/officeparser.browser.d.ts +79 -1
  10. package/dist/officeparser.browser.iife.js +112 -0
  11. package/dist/officeparser.browser.mjs +111 -0
  12. package/dist/parsers/ExcelParser.d.ts +1 -1
  13. package/dist/parsers/ExcelParser.js +71 -63
  14. package/dist/parsers/OpenOfficeParser.d.ts +1 -1
  15. package/dist/parsers/OpenOfficeParser.js +131 -114
  16. package/dist/parsers/PdfParser.d.ts +1 -1
  17. package/dist/parsers/PdfParser.js +98 -94
  18. package/dist/parsers/PowerPointParser.d.ts +1 -1
  19. package/dist/parsers/PowerPointParser.js +85 -88
  20. package/dist/parsers/RtfParser.d.ts +1 -1
  21. package/dist/parsers/RtfParser.js +10 -6
  22. package/dist/parsers/WordParser.d.ts +1 -1
  23. package/dist/parsers/WordParser.js +109 -101
  24. package/dist/sbom.cdx.json +1807 -0
  25. package/dist/types.d.ts +69 -1
  26. package/dist/utils/chartUtils.js +2 -0
  27. package/dist/utils/dateUtils.d.ts +17 -0
  28. package/dist/utils/dateUtils.js +69 -0
  29. package/dist/utils/envUtils.d.ts +24 -0
  30. package/dist/utils/envUtils.js +69 -0
  31. package/dist/utils/moduleLoader.d.ts +2 -1
  32. package/dist/utils/moduleLoader.js +9 -39
  33. package/dist/utils/ocrUtils.d.ts +16 -12
  34. package/dist/utils/ocrUtils.js +186 -25
  35. package/dist/utils/xmlUtils.d.ts +80 -9
  36. package/dist/utils/xmlUtils.js +236 -18
  37. package/dist/utils/zipUtils.js +6 -47
  38. package/package.json +31 -16
  39. package/dist/officeParserBundle@6.0.7.js +0 -154
  40. package/dist/officeparser.browser.js +0 -154
package/dist/types.d.ts CHANGED
@@ -1,3 +1,40 @@
1
+ /**
2
+ * Configuration options for OCR.
3
+ */
4
+ export interface OcrConfig {
5
+ /**
6
+ * Language for OCR.
7
+ * Default is 'eng'.
8
+ *
9
+ * You can provide multiple languages separated by a `+` sign (e.g., 'eng+fra' for English and French).
10
+ * The OCR engine will then attempt to recognize text in any of the specified languages.
11
+ *
12
+ * See the list of supported languages and their codes here:
13
+ * https://tesseract-ocr.github.io/tessdoc/Data-Files#data-files-for-version-400-november-29-2016
14
+ */
15
+ language?: string;
16
+ /**
17
+ * Path to the Tesseract worker script.
18
+ * Primarily used for offline/air-gapped environments.
19
+ */
20
+ workerPath?: string;
21
+ /**
22
+ * Path to the Tesseract core script.
23
+ * Primarily used for offline/air-gapped environments.
24
+ */
25
+ corePath?: string;
26
+ /**
27
+ * Path for Tesseract language files (traineddata).
28
+ * Primarily used for offline/air-gapped environments.
29
+ */
30
+ langPath?: string;
31
+ /**
32
+ * Timeout in milliseconds of inactivity before the OCR worker pool is automatically terminated.
33
+ * Set to 0 to disable auto-termination.
34
+ * Default is 10,000 (10 seconds).
35
+ */
36
+ autoTerminateTimeout?: number;
37
+ }
1
38
  /**
2
39
  * Configuration options for the OfficeParser.
3
40
  */
@@ -40,6 +77,7 @@ export interface OfficeParserConfig {
40
77
  */
41
78
  ocr?: boolean;
42
79
  /**
80
+ * @deprecated Use `ocrConfig.language` instead.
43
81
  * Language for OCR.
44
82
  * Default is 'eng'.
45
83
  *
@@ -50,11 +88,31 @@ export interface OfficeParserConfig {
50
88
  * https://tesseract-ocr.github.io/tessdoc/Data-Files#data-files-for-version-400-november-29-2016
51
89
  */
52
90
  ocrLanguage?: string;
91
+ /**
92
+ * Shared OCR configuration for worker pooling and offline support.
93
+ * If provided, `ocrLanguage` will be ignored in favor of `ocrConfig.language`.
94
+ */
95
+ ocrConfig?: OcrConfig;
96
+ /**
97
+ * Flag to serialize raw content (XML) as clean, formatted strings.
98
+ * Only relevant when `includeRawContent` is true.
99
+ * Default is true.
100
+ *
101
+ * If false, the parser will attempt to extract the original raw substring from the
102
+ * source document instead of re-serializing the DOM node.
103
+ */
104
+ serializeRawContent?: boolean;
105
+ /**
106
+ * Flag to preserve original XML whitespace and line endings when serializing.
107
+ * Only relevant when `includeRawContent` is true and `serializeRawContent` is true.
108
+ * Default is false.
109
+ */
110
+ preserveXmlWhitespace?: boolean;
53
111
  /**
54
112
  * The URL/path to the PDF.js worker script.
55
113
  *
56
114
  * **Mandatory** when using PDF parsing in browser environments to avoid worker configuration errors.
57
- * If not provided, it defaults to `https://unpkg.com/pdfjs-dist@5.5.207/build/pdf.worker.min.mjs`.
115
+ * If not provided, it defaults to `https://unpkg.com/pdfjs-dist@5.6.205/build/pdf.worker.min.mjs`.
58
116
  * You can override this with your own local path or a different CDN link.
59
117
  */
60
118
  pdfWorkerSrc?: string;
@@ -540,6 +598,16 @@ export interface OfficeMetadata {
540
598
  formatting?: Partial<TextFormatting>;
541
599
  /** Style map for styles in the document. */
542
600
  styleMap?: Record<string, Partial<TextFormatting>>;
601
+ /**
602
+ * User-defined custom properties embedded in the document.
603
+ * Sources by format:
604
+ * - DOCX/XLSX/PPTX: `docProps/custom.xml` (Office custom document properties)
605
+ * - ODT/ODP/ODS: `meta:user-defined` elements in `meta.xml`
606
+ * - PDF: non-standard entries in the PDF Info dictionary
607
+ * RTF does not support custom properties; the `\info` group is not extracted.
608
+ * Values are typed as string, number, boolean, or Date where the source format provides type information.
609
+ */
610
+ customProperties?: Record<string, string | number | boolean | Date>;
543
611
  }
544
612
  /**
545
613
  * The Abstract Syntax Tree (AST) returned by the parser.
@@ -43,6 +43,8 @@ const extractOpenXmlChartData = (xmlBuffer) => {
43
43
  const xml = xmlBuffer.toString("utf8");
44
44
  const dom = (0, xmlUtils_1.parseXmlString)(xml);
45
45
  const root = dom.documentElement;
46
+ if (!root)
47
+ return { title: undefined, xAxisTitle: undefined, yAxisTitle: undefined, dataSets: [], labels: [], rawTexts: [] };
46
48
  const title = extractOpenXmlRichText(root, "c:title");
47
49
  // Axis titles
48
50
  let xAxisTitle = undefined;
@@ -0,0 +1,17 @@
1
+ /**
2
+ * Date Parsing Utilities
3
+ *
4
+ * Provides robust functions for parsing date strings from various office formats.
5
+ * Handles standard ISO dates, PDF-specific date formats, and malformed strings.
6
+ *
7
+ * @module dateUtils
8
+ */
9
+ /**
10
+ * Parses a date string into a Date object.
11
+ * Handles standard ISO formats and falls back to native parsing.
12
+ * Returns undefined instead of "Invalid Date" if parsing fails.
13
+ *
14
+ * @param dateString - The date string to parse
15
+ * @returns Parsed Date object or undefined if parsing fails
16
+ */
17
+ export declare function parseOfficeDate(dateString: string | undefined): Date | undefined;
@@ -0,0 +1,69 @@
1
+ "use strict";
2
+ /**
3
+ * Date Parsing Utilities
4
+ *
5
+ * Provides robust functions for parsing date strings from various office formats.
6
+ * Handles standard ISO dates, PDF-specific date formats, and malformed strings.
7
+ *
8
+ * @module dateUtils
9
+ */
10
+ Object.defineProperty(exports, "__esModule", { value: true });
11
+ exports.parseOfficeDate = parseOfficeDate;
12
+ /**
13
+ * Parses a date string into a Date object.
14
+ * Handles standard ISO formats and falls back to native parsing.
15
+ * Returns undefined instead of "Invalid Date" if parsing fails.
16
+ *
17
+ * @param dateString - The date string to parse
18
+ * @returns Parsed Date object or undefined if parsing fails
19
+ */
20
+ function parseOfficeDate(dateString) {
21
+ if (!dateString)
22
+ return undefined;
23
+ try {
24
+ // PDF-specific format detection: D:YYYYMMDDHHmmSSOHH'mm'
25
+ if (dateString.startsWith('D:')) {
26
+ return parsePdfDate(dateString);
27
+ }
28
+ const date = new Date(dateString);
29
+ return isNaN(date.getTime()) ? undefined : date;
30
+ }
31
+ catch {
32
+ return undefined;
33
+ }
34
+ }
35
+ /**
36
+ * Internal helper for PDF-specific date format: D:YYYYMMDDHHmmSSOHH'mm'
37
+ * @param dateString - The PDF date string
38
+ */
39
+ function parsePdfDate(dateString) {
40
+ try {
41
+ // Remove "D:" prefix
42
+ let str = dateString.slice(2);
43
+ // Extract components: YYYYMMDDHHmmSS
44
+ const year = parseInt(str.slice(0, 4), 10);
45
+ const month = parseInt(str.slice(4, 6), 10) - 1; // 0-indexed
46
+ const day = parseInt(str.slice(6, 8), 10) || 1;
47
+ const hour = parseInt(str.slice(8, 10), 10) || 0;
48
+ const minute = parseInt(str.slice(10, 12), 10) || 0;
49
+ const second = parseInt(str.slice(12, 14), 10) || 0;
50
+ // Handle timezone if present
51
+ const tzMatch = str.slice(14).match(/([+-Z])(\d{2})'?(\d{2})?'?/);
52
+ if (tzMatch) {
53
+ if (tzMatch[1] === 'Z') {
54
+ return new Date(Date.UTC(year, month, day, hour, minute, second));
55
+ }
56
+ const tzSign = tzMatch[1] === '-' ? -1 : 1;
57
+ const tzHours = parseInt(tzMatch[2], 10) || 0;
58
+ const tzMinutes = parseInt(tzMatch[3], 10) || 0;
59
+ const offset = tzSign * (tzHours * 60 + tzMinutes);
60
+ // Create date in UTC and adjust for timezone
61
+ const utc = Date.UTC(year, month, day, hour, minute, second);
62
+ return new Date(utc - offset * 60000);
63
+ }
64
+ return new Date(year, month, day, hour, minute, second);
65
+ }
66
+ catch {
67
+ return undefined;
68
+ }
69
+ }
@@ -0,0 +1,24 @@
1
+ /**
2
+ * Environment detection and safe utility wrappers.
3
+ */
4
+ /**
5
+ * Detect if we are running in a browser environment.
6
+ */
7
+ export declare const isBrowser: boolean;
8
+ /**
9
+ * Supported Node.js-only features that require explicit guarding for browser compatibility.
10
+ */
11
+ export type NodeFeature = 'fs' | 'path-parsing' | 'pdf-worker-auto-resolution';
12
+ /**
13
+ * Throws an error if attempted to use Node.js-specific features in the browser.
14
+ *
15
+ * @param feature - The Node.js feature being accessed
16
+ * @throws {Error} Clear error message directing browser users to use Buffers
17
+ */
18
+ export declare function assertNode(feature: NodeFeature): void;
19
+ /**
20
+ * Polyfills DOMMatrix if not available globally (required for Node.js < 20).
21
+ * This shim provides enough properties for pdfjs-dist 5.x to calculate
22
+ * text coordinates and transformations.
23
+ */
24
+ export declare function ensureDomMatrix(): void;
@@ -0,0 +1,69 @@
1
+ "use strict";
2
+ /**
3
+ * Environment detection and safe utility wrappers.
4
+ */
5
+ Object.defineProperty(exports, "__esModule", { value: true });
6
+ exports.isBrowser = void 0;
7
+ exports.assertNode = assertNode;
8
+ exports.ensureDomMatrix = ensureDomMatrix;
9
+ /**
10
+ * Detect if we are running in a browser environment.
11
+ */
12
+ exports.isBrowser = typeof window !== 'undefined' && typeof window.document !== 'undefined';
13
+ /**
14
+ * Human-readable descriptions for Node-only features.
15
+ */
16
+ const readableFeatures = {
17
+ 'fs': 'direct file system access',
18
+ 'path-parsing': 'parsing from file path string',
19
+ 'pdf-worker-auto-resolution': 'automatic PDF worker resolution from node_modules'
20
+ };
21
+ /**
22
+ * Throws an error if attempted to use Node.js-specific features in the browser.
23
+ *
24
+ * @param feature - The Node.js feature being accessed
25
+ * @throws {Error} Clear error message directing browser users to use Buffers
26
+ */
27
+ function assertNode(feature) {
28
+ if (exports.isBrowser) {
29
+ throw new Error(`officeparser: '${readableFeatures[feature]}' is not supported in the browser. Browser users must pass file content as Buffer or ArrayBuffer directly.`);
30
+ }
31
+ }
32
+ /**
33
+ * Polyfills DOMMatrix if not available globally (required for Node.js < 20).
34
+ * This shim provides enough properties for pdfjs-dist 5.x to calculate
35
+ * text coordinates and transformations.
36
+ */
37
+ function ensureDomMatrix() {
38
+ if (typeof global !== 'undefined' && !global.DOMMatrix) {
39
+ global.DOMMatrix = class DOMMatrix {
40
+ a;
41
+ b;
42
+ c;
43
+ d;
44
+ e;
45
+ f;
46
+ constructor(init) {
47
+ if (Array.isArray(init) && init.length >= 6) {
48
+ this.a = init[0];
49
+ this.b = init[1];
50
+ this.c = init[2];
51
+ this.d = init[3];
52
+ this.e = init[4];
53
+ this.f = init[5];
54
+ }
55
+ else {
56
+ this.a = this.d = 1;
57
+ this.b = this.c = this.e = this.f = 0;
58
+ }
59
+ }
60
+ // Map standard matrix properties for compatibility
61
+ get m11() { return this.a; }
62
+ get m12() { return this.b; }
63
+ get m21() { return this.c; }
64
+ get m22() { return this.d; }
65
+ get m41() { return this.e; }
66
+ get m42() { return this.f; }
67
+ };
68
+ }
69
+ }
@@ -7,10 +7,11 @@
7
7
  * This approach prevents TypeScript from transpiling dynamic import()
8
8
  * into require() when targeting CommonJS.
9
9
  */
10
+ import type * as FileTypeModule from 'file-type' with { 'resolution-mode': 'import' };
10
11
  /**
11
12
  * Specialized loader for file-type
12
13
  */
13
- export declare function loadFileType(): Promise<typeof import('file-type')>;
14
+ export declare function loadFileType(): Promise<typeof FileTypeModule>;
14
15
  /**
15
16
  * Specialized loader for pdfjs-dist
16
17
  */
@@ -8,42 +8,10 @@
8
8
  * This approach prevents TypeScript from transpiling dynamic import()
9
9
  * into require() when targeting CommonJS.
10
10
  */
11
- var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
12
- if (k2 === undefined) k2 = k;
13
- var desc = Object.getOwnPropertyDescriptor(m, k);
14
- if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
15
- desc = { enumerable: true, get: function() { return m[k]; } };
16
- }
17
- Object.defineProperty(o, k2, desc);
18
- }) : (function(o, m, k, k2) {
19
- if (k2 === undefined) k2 = k;
20
- o[k2] = m[k];
21
- }));
22
- var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
23
- Object.defineProperty(o, "default", { enumerable: true, value: v });
24
- }) : function(o, v) {
25
- o["default"] = v;
26
- });
27
- var __importStar = (this && this.__importStar) || (function () {
28
- var ownKeys = function(o) {
29
- ownKeys = Object.getOwnPropertyNames || function (o) {
30
- var ar = [];
31
- for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
32
- return ar;
33
- };
34
- return ownKeys(o);
35
- };
36
- return function (mod) {
37
- if (mod && mod.__esModule) return mod;
38
- var result = {};
39
- if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
40
- __setModuleDefault(result, mod);
41
- return result;
42
- };
43
- })();
44
11
  Object.defineProperty(exports, "__esModule", { value: true });
45
12
  exports.loadFileType = loadFileType;
46
13
  exports.loadPdfJs = loadPdfJs;
14
+ const envUtils_js_1 = require("./envUtils.js");
47
15
  /**
48
16
  * Dynamically loads an ESM module in a Node.js CJS context.
49
17
  *
@@ -59,18 +27,20 @@ async function loadNodeEsmModule(specifier) {
59
27
  * Specialized loader for file-type
60
28
  */
61
29
  async function loadFileType() {
62
- if (typeof window === 'undefined') {
63
- // Node.js path
30
+ if (!envUtils_js_1.isBrowser) {
31
+ // Node.js path: Use dynamic import wrapper for CJS compatibility
64
32
  return loadNodeEsmModule('file-type');
65
33
  }
66
- // Browser path: esbuild handles standard dynamic import()
67
- return Promise.resolve().then(() => __importStar(require('file-type')));
34
+ // Browser path: standard dynamic import() is handled by bundlers (e.g. esbuild/Vite)
35
+ return import('file-type');
68
36
  }
69
37
  /**
70
38
  * Specialized loader for pdfjs-dist
71
39
  */
72
40
  async function loadPdfJs() {
73
- if (typeof window === 'undefined') {
41
+ if (!envUtils_js_1.isBrowser) {
42
+ // Ensure DOMMatrix polyfill for Node.js 18 support
43
+ (0, envUtils_js_1.ensureDomMatrix)();
74
44
  // Node.js environment: require legacy build for stability with ESM-only main
75
45
  try {
76
46
  return await loadNodeEsmModule('pdfjs-dist/legacy/build/pdf.mjs');
@@ -80,5 +50,5 @@ async function loadPdfJs() {
80
50
  }
81
51
  }
82
52
  // Browser environment: esbuild handles standard static-looking dynamic import()
83
- return Promise.resolve().then(() => __importStar(require('pdfjs-dist')));
53
+ return import('pdfjs-dist');
84
54
  }
@@ -4,8 +4,12 @@
4
4
  * This module provides functions for extracting text from images using Tesseract.js.
5
5
  * Used when `config.ocr` is enabled to extract text from embedded images in documents.
6
6
  *
7
+ * Includes a worker pool via OcrSchedulerManager to improve performance when
8
+ * processing multiple images.
9
+ *
7
10
  * @module ocrUtils
8
11
  */
12
+ import { OcrConfig } from '../types.js';
9
13
  /**
10
14
  * Performs Optical Character Recognition (OCR) on an image to extract text.
11
15
  *
@@ -13,26 +17,26 @@
13
17
  * This is useful for extracting text from screenshots, scanned documents,
14
18
  * charts with labels, or any image containing text.
15
19
  *
16
- * The function creates a new Tesseract worker, processes the image,
17
- * and properly terminates the worker to free resources.
20
+ * This function uses a shared worker pool to minimize initialization overhead.
18
21
  *
19
- * @param imageBuffer - The image data as a Node.js Buffer (PNG, JPEG, etc.)
20
- * @param language - The language code for OCR (default: 'eng' for English).
21
- * Supports ISO 639-2/T three-letter codes: 'eng', 'spa', 'fra', 'deu', etc.
22
- * Multiple languages can be combined with '+': 'eng+fra'
22
+ * @param image - The image data as a Buffer, file path, or Blob
23
+ * @param config - Optional configuration for language and custom worker paths
23
24
  * @returns A promise that resolves to the recognized text as a string
24
25
  * @throws {Error} If the image cannot be processed or Tesseract initialization fails
25
26
  *
26
27
  * @example
27
28
  * ```typescript
28
29
  * // Extract text from an English image
29
- * const text = await performOcr(imageBuffer, 'eng');
30
- * console.log(text); // "Annual Revenue: $1.2M"
31
- *
32
- * // Extract text from a multilingual image
33
- * const text = await performOcr(imageBuffer, 'eng+spa');
30
+ * const text = await performOcr(imageBuffer, { language: 'eng' });
34
31
  * ```
35
32
  *
36
33
  * @see https://github.com/naptha/tesseract.js for supported languages and options
37
34
  */
38
- export declare const performOcr: (image: Buffer | string, language?: string) => Promise<string>;
35
+ export declare const performOcr: (image: Buffer | string, config?: OcrConfig) => Promise<string>;
36
+ /**
37
+ * Terminates all OCR workers and cleans up resources.
38
+ *
39
+ * Should be called when the application is shutting down or OCR is no longer needed
40
+ * to prevent memory leaks and dangling worker processes.
41
+ */
42
+ export declare const terminateOcr: () => Promise<void>;
@@ -5,11 +5,176 @@
5
5
  * This module provides functions for extracting text from images using Tesseract.js.
6
6
  * Used when `config.ocr` is enabled to extract text from embedded images in documents.
7
7
  *
8
+ * Includes a worker pool via OcrSchedulerManager to improve performance when
9
+ * processing multiple images.
10
+ *
8
11
  * @module ocrUtils
9
12
  */
10
13
  Object.defineProperty(exports, "__esModule", { value: true });
11
- exports.performOcr = void 0;
12
- const tesseract_js_1 = require("tesseract.js");
14
+ exports.terminateOcr = exports.performOcr = void 0;
15
+ /**
16
+ * Manages a pool of Tesseract workers with "Smart Affinity".
17
+ *
18
+ * Instead of a simple scheduler, this manager allows workers to persist with
19
+ * a specific language affinity. If a new language is requested and the pool
20
+ * is at capacity, it re-initializes the Least Recently Used (LRU) idle worker
21
+ * rather than resetting the entire pool.
22
+ *
23
+ * Implements lazy loading of tesseract.js to ensure no background processes
24
+ * are spawned unless OCR is explicitly used.
25
+ */
26
+ class OcrSchedulerManager {
27
+ static instance;
28
+ pool = [];
29
+ queue = [];
30
+ MAX_WORKERS = 4;
31
+ idleTimeout = 10000; // 10s default
32
+ timeoutId = null;
33
+ constructor() { }
34
+ /**
35
+ * Returns the singleton instance of the manager.
36
+ */
37
+ static getInstance() {
38
+ if (!OcrSchedulerManager.instance) {
39
+ OcrSchedulerManager.instance = new OcrSchedulerManager();
40
+ }
41
+ return OcrSchedulerManager.instance;
42
+ }
43
+ /**
44
+ * Checks if the singleton instance has been initialized.
45
+ */
46
+ static hasInstance() {
47
+ return !!OcrSchedulerManager.instance;
48
+ }
49
+ /**
50
+ * Resets the inactivity timer. If the timer reaches its duration,
51
+ * all workers are terminated automatically.
52
+ */
53
+ resetIdleTimer() {
54
+ if (this.timeoutId) {
55
+ clearTimeout(this.timeoutId);
56
+ }
57
+ if (this.idleTimeout > 0) {
58
+ this.timeoutId = setTimeout(async () => {
59
+ await this.terminate();
60
+ }, this.idleTimeout);
61
+ }
62
+ }
63
+ /**
64
+ * Performs OCR on an image using the smart worker pool.
65
+ *
66
+ * @param image - Image data (Buffer, string path, or Blob)
67
+ * @param config - OCR configuration (language, custom paths)
68
+ * @returns Recognized text
69
+ */
70
+ async recognize(image, config) {
71
+ return new Promise((resolve, reject) => {
72
+ // Update idle timeout if provided
73
+ if (config?.autoTerminateTimeout !== undefined) {
74
+ this.idleTimeout = config.autoTerminateTimeout;
75
+ }
76
+ // Reset the inactivity timer every time a new job is requested
77
+ this.resetIdleTimer();
78
+ // Add job to queue and trigger processing
79
+ this.queue.push({ image, config: config || {}, resolve, reject });
80
+ this.processQueue();
81
+ });
82
+ }
83
+ /**
84
+ * Attempts to process the next job in the queue using an available worker.
85
+ */
86
+ async processQueue() {
87
+ if (this.queue.length === 0)
88
+ return;
89
+ const nextJob = this.queue[0];
90
+ const requestedLanguage = nextJob.config.language || 'eng';
91
+ // 1. Find an idle worker with the EXACT language affinity
92
+ let managed = this.pool.find(mw => !mw.isBusy && mw.language === requestedLanguage);
93
+ // 2. If not found and we have room, create a new worker
94
+ if (!managed && this.pool.length < this.MAX_WORKERS) {
95
+ try {
96
+ const { createWorker } = await import('tesseract.js');
97
+ const options = { logger: () => { } };
98
+ if (nextJob.config.workerPath)
99
+ options.workerPath = nextJob.config.workerPath;
100
+ if (nextJob.config.corePath)
101
+ options.corePath = nextJob.config.corePath;
102
+ if (nextJob.config.langPath)
103
+ options.langPath = nextJob.config.langPath;
104
+ const worker = await createWorker(requestedLanguage, 1, options);
105
+ managed = {
106
+ worker,
107
+ language: requestedLanguage,
108
+ lastUsed: Date.now(),
109
+ isBusy: false
110
+ };
111
+ this.pool.push(managed);
112
+ }
113
+ catch (err) {
114
+ const job = this.queue.shift();
115
+ job?.reject(err);
116
+ this.processQueue(); // Try next job
117
+ return;
118
+ }
119
+ }
120
+ // 3. If still not found and we are at capacity, find the LRU idle worker and re-initialize it
121
+ if (!managed) {
122
+ const idleWorkers = this.pool.filter(mw => !mw.isBusy);
123
+ if (idleWorkers.length > 0) {
124
+ // Find Least Recently Used idle worker
125
+ managed = idleWorkers.reduce((prev, curr) => (prev.lastUsed < curr.lastUsed ? prev : curr));
126
+ try {
127
+ // Smart Re-initialization (v5 API)
128
+ await managed.worker.reinitialize(requestedLanguage);
129
+ managed.language = requestedLanguage;
130
+ }
131
+ catch (err) {
132
+ // If reinitialization fails, we might need to recreate it, but for simplicity
133
+ // we'll just fail this job and try another worker next time.
134
+ const job = this.queue.shift();
135
+ job?.reject(err);
136
+ this.processQueue();
137
+ return;
138
+ }
139
+ }
140
+ }
141
+ // 4. If we have a worker ready, execute the job
142
+ if (managed) {
143
+ const job = this.queue.shift();
144
+ if (!job)
145
+ return;
146
+ managed.isBusy = true;
147
+ managed.lastUsed = Date.now();
148
+ try {
149
+ const { data: { text } } = await managed.worker.recognize(job.image);
150
+ job.resolve(text);
151
+ }
152
+ catch (err) {
153
+ job.reject(err);
154
+ }
155
+ finally {
156
+ managed.isBusy = false;
157
+ managed.lastUsed = Date.now();
158
+ // Check if there are more jobs waiting
159
+ this.processQueue();
160
+ }
161
+ }
162
+ // If no worker is available (all busy), the job stays in the queue
163
+ // and will be picked up when a worker finishes.
164
+ }
165
+ /**
166
+ * Terminates all workers in the pool and resets the state.
167
+ */
168
+ async terminate() {
169
+ if (this.timeoutId) {
170
+ clearTimeout(this.timeoutId);
171
+ this.timeoutId = null;
172
+ }
173
+ const workersToTerminate = this.pool.map(mw => mw.worker.terminate());
174
+ await Promise.all(workersToTerminate);
175
+ this.pool = [];
176
+ }
177
+ }
13
178
  /**
14
179
  * Performs Optical Character Recognition (OCR) on an image to extract text.
15
180
  *
@@ -17,45 +182,41 @@ const tesseract_js_1 = require("tesseract.js");
17
182
  * This is useful for extracting text from screenshots, scanned documents,
18
183
  * charts with labels, or any image containing text.
19
184
  *
20
- * The function creates a new Tesseract worker, processes the image,
21
- * and properly terminates the worker to free resources.
185
+ * This function uses a shared worker pool to minimize initialization overhead.
22
186
  *
23
- * @param imageBuffer - The image data as a Node.js Buffer (PNG, JPEG, etc.)
24
- * @param language - The language code for OCR (default: 'eng' for English).
25
- * Supports ISO 639-2/T three-letter codes: 'eng', 'spa', 'fra', 'deu', etc.
26
- * Multiple languages can be combined with '+': 'eng+fra'
187
+ * @param image - The image data as a Buffer, file path, or Blob
188
+ * @param config - Optional configuration for language and custom worker paths
27
189
  * @returns A promise that resolves to the recognized text as a string
28
190
  * @throws {Error} If the image cannot be processed or Tesseract initialization fails
29
191
  *
30
192
  * @example
31
193
  * ```typescript
32
194
  * // Extract text from an English image
33
- * const text = await performOcr(imageBuffer, 'eng');
34
- * console.log(text); // "Annual Revenue: $1.2M"
35
- *
36
- * // Extract text from a multilingual image
37
- * const text = await performOcr(imageBuffer, 'eng+spa');
195
+ * const text = await performOcr(imageBuffer, { language: 'eng' });
38
196
  * ```
39
197
  *
40
198
  * @see https://github.com/naptha/tesseract.js for supported languages and options
41
199
  */
42
- const performOcr = async (image, language = 'eng') => {
43
- // Step 1: Create a Tesseract worker with the specified language
44
- // We pass 1 for OEM (LSTM) and a silent logger to suppress console output
45
- const worker = await (0, tesseract_js_1.createWorker)(language, 1, {
46
- logger: () => { }
47
- });
48
- // Step 2: Prepare image data
200
+ const performOcr = async (image, config) => {
201
+ // Prepare image data
49
202
  let inputImage = image;
50
203
  // In browser environment, convert Buffer to Blob for better compatibility
51
204
  // @ts-ignore
52
205
  if (typeof window !== 'undefined' && typeof Blob !== 'undefined' && Buffer.isBuffer(image)) {
53
206
  inputImage = new Blob([image], { type: 'image/bmp' });
54
207
  }
55
- // Step 3: Perform OCR
56
- const ret = await worker.recognize(inputImage);
57
- // Step 4: Terminate worker
58
- await worker.terminate();
59
- return ret.data.text;
208
+ return await OcrSchedulerManager.getInstance().recognize(inputImage, config);
60
209
  };
61
210
  exports.performOcr = performOcr;
211
+ /**
212
+ * Terminates all OCR workers and cleans up resources.
213
+ *
214
+ * Should be called when the application is shutting down or OCR is no longer needed
215
+ * to prevent memory leaks and dangling worker processes.
216
+ */
217
+ const terminateOcr = async () => {
218
+ if (OcrSchedulerManager.hasInstance()) {
219
+ await OcrSchedulerManager.getInstance().terminate();
220
+ }
221
+ };
222
+ exports.terminateOcr = terminateOcr;