officeparser 6.0.7 → 6.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +92 -13
- package/dist/OfficeParser.d.ts +10 -1
- package/dist/OfficeParser.js +43 -56
- package/dist/cli.d.ts +20 -0
- package/dist/cli.js +116 -0
- package/dist/index.d.ts +3 -3
- package/dist/index.js +7 -59
- package/dist/index.mjs +18 -0
- package/dist/officeparser.browser.d.ts +79 -1
- package/dist/officeparser.browser.iife.js +112 -0
- package/dist/officeparser.browser.mjs +111 -0
- package/dist/parsers/ExcelParser.d.ts +1 -1
- package/dist/parsers/ExcelParser.js +71 -63
- package/dist/parsers/OpenOfficeParser.d.ts +1 -1
- package/dist/parsers/OpenOfficeParser.js +131 -114
- package/dist/parsers/PdfParser.d.ts +1 -1
- package/dist/parsers/PdfParser.js +98 -94
- package/dist/parsers/PowerPointParser.d.ts +1 -1
- package/dist/parsers/PowerPointParser.js +85 -88
- package/dist/parsers/RtfParser.d.ts +1 -1
- package/dist/parsers/RtfParser.js +10 -6
- package/dist/parsers/WordParser.d.ts +1 -1
- package/dist/parsers/WordParser.js +109 -101
- package/dist/sbom.cdx.json +1807 -0
- package/dist/types.d.ts +69 -1
- package/dist/utils/chartUtils.js +2 -0
- package/dist/utils/dateUtils.d.ts +17 -0
- package/dist/utils/dateUtils.js +69 -0
- package/dist/utils/envUtils.d.ts +24 -0
- package/dist/utils/envUtils.js +69 -0
- package/dist/utils/moduleLoader.d.ts +2 -1
- package/dist/utils/moduleLoader.js +9 -39
- package/dist/utils/ocrUtils.d.ts +16 -12
- package/dist/utils/ocrUtils.js +186 -25
- package/dist/utils/xmlUtils.d.ts +80 -9
- package/dist/utils/xmlUtils.js +236 -18
- package/dist/utils/zipUtils.js +6 -47
- package/package.json +31 -16
- package/dist/officeParserBundle@6.0.7.js +0 -154
- package/dist/officeparser.browser.js +0 -154
package/dist/types.d.ts
CHANGED
|
@@ -1,3 +1,40 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Configuration options for OCR.
|
|
3
|
+
*/
|
|
4
|
+
export interface OcrConfig {
|
|
5
|
+
/**
|
|
6
|
+
* Language for OCR.
|
|
7
|
+
* Default is 'eng'.
|
|
8
|
+
*
|
|
9
|
+
* You can provide multiple languages separated by a `+` sign (e.g., 'eng+fra' for English and French).
|
|
10
|
+
* The OCR engine will then attempt to recognize text in any of the specified languages.
|
|
11
|
+
*
|
|
12
|
+
* See the list of supported languages and their codes here:
|
|
13
|
+
* https://tesseract-ocr.github.io/tessdoc/Data-Files#data-files-for-version-400-november-29-2016
|
|
14
|
+
*/
|
|
15
|
+
language?: string;
|
|
16
|
+
/**
|
|
17
|
+
* Path to the Tesseract worker script.
|
|
18
|
+
* Primarily used for offline/air-gapped environments.
|
|
19
|
+
*/
|
|
20
|
+
workerPath?: string;
|
|
21
|
+
/**
|
|
22
|
+
* Path to the Tesseract core script.
|
|
23
|
+
* Primarily used for offline/air-gapped environments.
|
|
24
|
+
*/
|
|
25
|
+
corePath?: string;
|
|
26
|
+
/**
|
|
27
|
+
* Path for Tesseract language files (traineddata).
|
|
28
|
+
* Primarily used for offline/air-gapped environments.
|
|
29
|
+
*/
|
|
30
|
+
langPath?: string;
|
|
31
|
+
/**
|
|
32
|
+
* Timeout in milliseconds of inactivity before the OCR worker pool is automatically terminated.
|
|
33
|
+
* Set to 0 to disable auto-termination.
|
|
34
|
+
* Default is 10,000 (10 seconds).
|
|
35
|
+
*/
|
|
36
|
+
autoTerminateTimeout?: number;
|
|
37
|
+
}
|
|
1
38
|
/**
|
|
2
39
|
* Configuration options for the OfficeParser.
|
|
3
40
|
*/
|
|
@@ -40,6 +77,7 @@ export interface OfficeParserConfig {
|
|
|
40
77
|
*/
|
|
41
78
|
ocr?: boolean;
|
|
42
79
|
/**
|
|
80
|
+
* @deprecated Use `ocrConfig.language` instead.
|
|
43
81
|
* Language for OCR.
|
|
44
82
|
* Default is 'eng'.
|
|
45
83
|
*
|
|
@@ -50,11 +88,31 @@ export interface OfficeParserConfig {
|
|
|
50
88
|
* https://tesseract-ocr.github.io/tessdoc/Data-Files#data-files-for-version-400-november-29-2016
|
|
51
89
|
*/
|
|
52
90
|
ocrLanguage?: string;
|
|
91
|
+
/**
|
|
92
|
+
* Shared OCR configuration for worker pooling and offline support.
|
|
93
|
+
* If provided, `ocrLanguage` will be ignored in favor of `ocrConfig.language`.
|
|
94
|
+
*/
|
|
95
|
+
ocrConfig?: OcrConfig;
|
|
96
|
+
/**
|
|
97
|
+
* Flag to serialize raw content (XML) as clean, formatted strings.
|
|
98
|
+
* Only relevant when `includeRawContent` is true.
|
|
99
|
+
* Default is true.
|
|
100
|
+
*
|
|
101
|
+
* If false, the parser will attempt to extract the original raw substring from the
|
|
102
|
+
* source document instead of re-serializing the DOM node.
|
|
103
|
+
*/
|
|
104
|
+
serializeRawContent?: boolean;
|
|
105
|
+
/**
|
|
106
|
+
* Flag to preserve original XML whitespace and line endings when serializing.
|
|
107
|
+
* Only relevant when `includeRawContent` is true and `serializeRawContent` is true.
|
|
108
|
+
* Default is false.
|
|
109
|
+
*/
|
|
110
|
+
preserveXmlWhitespace?: boolean;
|
|
53
111
|
/**
|
|
54
112
|
* The URL/path to the PDF.js worker script.
|
|
55
113
|
*
|
|
56
114
|
* **Mandatory** when using PDF parsing in browser environments to avoid worker configuration errors.
|
|
57
|
-
* If not provided, it defaults to `https://unpkg.com/pdfjs-dist@5.
|
|
115
|
+
* If not provided, it defaults to `https://unpkg.com/pdfjs-dist@5.6.205/build/pdf.worker.min.mjs`.
|
|
58
116
|
* You can override this with your own local path or a different CDN link.
|
|
59
117
|
*/
|
|
60
118
|
pdfWorkerSrc?: string;
|
|
@@ -540,6 +598,16 @@ export interface OfficeMetadata {
|
|
|
540
598
|
formatting?: Partial<TextFormatting>;
|
|
541
599
|
/** Style map for styles in the document. */
|
|
542
600
|
styleMap?: Record<string, Partial<TextFormatting>>;
|
|
601
|
+
/**
|
|
602
|
+
* User-defined custom properties embedded in the document.
|
|
603
|
+
* Sources by format:
|
|
604
|
+
* - DOCX/XLSX/PPTX: `docProps/custom.xml` (Office custom document properties)
|
|
605
|
+
* - ODT/ODP/ODS: `meta:user-defined` elements in `meta.xml`
|
|
606
|
+
* - PDF: non-standard entries in the PDF Info dictionary
|
|
607
|
+
* RTF does not support custom properties; the `\info` group is not extracted.
|
|
608
|
+
* Values are typed as string, number, boolean, or Date where the source format provides type information.
|
|
609
|
+
*/
|
|
610
|
+
customProperties?: Record<string, string | number | boolean | Date>;
|
|
543
611
|
}
|
|
544
612
|
/**
|
|
545
613
|
* The Abstract Syntax Tree (AST) returned by the parser.
|
package/dist/utils/chartUtils.js
CHANGED
|
@@ -43,6 +43,8 @@ const extractOpenXmlChartData = (xmlBuffer) => {
|
|
|
43
43
|
const xml = xmlBuffer.toString("utf8");
|
|
44
44
|
const dom = (0, xmlUtils_1.parseXmlString)(xml);
|
|
45
45
|
const root = dom.documentElement;
|
|
46
|
+
if (!root)
|
|
47
|
+
return { title: undefined, xAxisTitle: undefined, yAxisTitle: undefined, dataSets: [], labels: [], rawTexts: [] };
|
|
46
48
|
const title = extractOpenXmlRichText(root, "c:title");
|
|
47
49
|
// Axis titles
|
|
48
50
|
let xAxisTitle = undefined;
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Date Parsing Utilities
|
|
3
|
+
*
|
|
4
|
+
* Provides robust functions for parsing date strings from various office formats.
|
|
5
|
+
* Handles standard ISO dates, PDF-specific date formats, and malformed strings.
|
|
6
|
+
*
|
|
7
|
+
* @module dateUtils
|
|
8
|
+
*/
|
|
9
|
+
/**
|
|
10
|
+
* Parses a date string into a Date object.
|
|
11
|
+
* Handles standard ISO formats and falls back to native parsing.
|
|
12
|
+
* Returns undefined instead of "Invalid Date" if parsing fails.
|
|
13
|
+
*
|
|
14
|
+
* @param dateString - The date string to parse
|
|
15
|
+
* @returns Parsed Date object or undefined if parsing fails
|
|
16
|
+
*/
|
|
17
|
+
export declare function parseOfficeDate(dateString: string | undefined): Date | undefined;
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* Date Parsing Utilities
|
|
4
|
+
*
|
|
5
|
+
* Provides robust functions for parsing date strings from various office formats.
|
|
6
|
+
* Handles standard ISO dates, PDF-specific date formats, and malformed strings.
|
|
7
|
+
*
|
|
8
|
+
* @module dateUtils
|
|
9
|
+
*/
|
|
10
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
11
|
+
exports.parseOfficeDate = parseOfficeDate;
|
|
12
|
+
/**
|
|
13
|
+
* Parses a date string into a Date object.
|
|
14
|
+
* Handles standard ISO formats and falls back to native parsing.
|
|
15
|
+
* Returns undefined instead of "Invalid Date" if parsing fails.
|
|
16
|
+
*
|
|
17
|
+
* @param dateString - The date string to parse
|
|
18
|
+
* @returns Parsed Date object or undefined if parsing fails
|
|
19
|
+
*/
|
|
20
|
+
function parseOfficeDate(dateString) {
|
|
21
|
+
if (!dateString)
|
|
22
|
+
return undefined;
|
|
23
|
+
try {
|
|
24
|
+
// PDF-specific format detection: D:YYYYMMDDHHmmSSOHH'mm'
|
|
25
|
+
if (dateString.startsWith('D:')) {
|
|
26
|
+
return parsePdfDate(dateString);
|
|
27
|
+
}
|
|
28
|
+
const date = new Date(dateString);
|
|
29
|
+
return isNaN(date.getTime()) ? undefined : date;
|
|
30
|
+
}
|
|
31
|
+
catch {
|
|
32
|
+
return undefined;
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
/**
|
|
36
|
+
* Internal helper for PDF-specific date format: D:YYYYMMDDHHmmSSOHH'mm'
|
|
37
|
+
* @param dateString - The PDF date string
|
|
38
|
+
*/
|
|
39
|
+
function parsePdfDate(dateString) {
|
|
40
|
+
try {
|
|
41
|
+
// Remove "D:" prefix
|
|
42
|
+
let str = dateString.slice(2);
|
|
43
|
+
// Extract components: YYYYMMDDHHmmSS
|
|
44
|
+
const year = parseInt(str.slice(0, 4), 10);
|
|
45
|
+
const month = parseInt(str.slice(4, 6), 10) - 1; // 0-indexed
|
|
46
|
+
const day = parseInt(str.slice(6, 8), 10) || 1;
|
|
47
|
+
const hour = parseInt(str.slice(8, 10), 10) || 0;
|
|
48
|
+
const minute = parseInt(str.slice(10, 12), 10) || 0;
|
|
49
|
+
const second = parseInt(str.slice(12, 14), 10) || 0;
|
|
50
|
+
// Handle timezone if present
|
|
51
|
+
const tzMatch = str.slice(14).match(/([+-Z])(\d{2})'?(\d{2})?'?/);
|
|
52
|
+
if (tzMatch) {
|
|
53
|
+
if (tzMatch[1] === 'Z') {
|
|
54
|
+
return new Date(Date.UTC(year, month, day, hour, minute, second));
|
|
55
|
+
}
|
|
56
|
+
const tzSign = tzMatch[1] === '-' ? -1 : 1;
|
|
57
|
+
const tzHours = parseInt(tzMatch[2], 10) || 0;
|
|
58
|
+
const tzMinutes = parseInt(tzMatch[3], 10) || 0;
|
|
59
|
+
const offset = tzSign * (tzHours * 60 + tzMinutes);
|
|
60
|
+
// Create date in UTC and adjust for timezone
|
|
61
|
+
const utc = Date.UTC(year, month, day, hour, minute, second);
|
|
62
|
+
return new Date(utc - offset * 60000);
|
|
63
|
+
}
|
|
64
|
+
return new Date(year, month, day, hour, minute, second);
|
|
65
|
+
}
|
|
66
|
+
catch {
|
|
67
|
+
return undefined;
|
|
68
|
+
}
|
|
69
|
+
}
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Environment detection and safe utility wrappers.
|
|
3
|
+
*/
|
|
4
|
+
/**
|
|
5
|
+
* Detect if we are running in a browser environment.
|
|
6
|
+
*/
|
|
7
|
+
export declare const isBrowser: boolean;
|
|
8
|
+
/**
|
|
9
|
+
* Supported Node.js-only features that require explicit guarding for browser compatibility.
|
|
10
|
+
*/
|
|
11
|
+
export type NodeFeature = 'fs' | 'path-parsing' | 'pdf-worker-auto-resolution';
|
|
12
|
+
/**
|
|
13
|
+
* Throws an error if attempted to use Node.js-specific features in the browser.
|
|
14
|
+
*
|
|
15
|
+
* @param feature - The Node.js feature being accessed
|
|
16
|
+
* @throws {Error} Clear error message directing browser users to use Buffers
|
|
17
|
+
*/
|
|
18
|
+
export declare function assertNode(feature: NodeFeature): void;
|
|
19
|
+
/**
|
|
20
|
+
* Polyfills DOMMatrix if not available globally (required for Node.js < 20).
|
|
21
|
+
* This shim provides enough properties for pdfjs-dist 5.x to calculate
|
|
22
|
+
* text coordinates and transformations.
|
|
23
|
+
*/
|
|
24
|
+
export declare function ensureDomMatrix(): void;
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* Environment detection and safe utility wrappers.
|
|
4
|
+
*/
|
|
5
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
6
|
+
exports.isBrowser = void 0;
|
|
7
|
+
exports.assertNode = assertNode;
|
|
8
|
+
exports.ensureDomMatrix = ensureDomMatrix;
|
|
9
|
+
/**
|
|
10
|
+
* Detect if we are running in a browser environment.
|
|
11
|
+
*/
|
|
12
|
+
exports.isBrowser = typeof window !== 'undefined' && typeof window.document !== 'undefined';
|
|
13
|
+
/**
|
|
14
|
+
* Human-readable descriptions for Node-only features.
|
|
15
|
+
*/
|
|
16
|
+
const readableFeatures = {
|
|
17
|
+
'fs': 'direct file system access',
|
|
18
|
+
'path-parsing': 'parsing from file path string',
|
|
19
|
+
'pdf-worker-auto-resolution': 'automatic PDF worker resolution from node_modules'
|
|
20
|
+
};
|
|
21
|
+
/**
|
|
22
|
+
* Throws an error if attempted to use Node.js-specific features in the browser.
|
|
23
|
+
*
|
|
24
|
+
* @param feature - The Node.js feature being accessed
|
|
25
|
+
* @throws {Error} Clear error message directing browser users to use Buffers
|
|
26
|
+
*/
|
|
27
|
+
function assertNode(feature) {
|
|
28
|
+
if (exports.isBrowser) {
|
|
29
|
+
throw new Error(`officeparser: '${readableFeatures[feature]}' is not supported in the browser. Browser users must pass file content as Buffer or ArrayBuffer directly.`);
|
|
30
|
+
}
|
|
31
|
+
}
|
|
32
|
+
/**
|
|
33
|
+
* Polyfills DOMMatrix if not available globally (required for Node.js < 20).
|
|
34
|
+
* This shim provides enough properties for pdfjs-dist 5.x to calculate
|
|
35
|
+
* text coordinates and transformations.
|
|
36
|
+
*/
|
|
37
|
+
function ensureDomMatrix() {
|
|
38
|
+
if (typeof global !== 'undefined' && !global.DOMMatrix) {
|
|
39
|
+
global.DOMMatrix = class DOMMatrix {
|
|
40
|
+
a;
|
|
41
|
+
b;
|
|
42
|
+
c;
|
|
43
|
+
d;
|
|
44
|
+
e;
|
|
45
|
+
f;
|
|
46
|
+
constructor(init) {
|
|
47
|
+
if (Array.isArray(init) && init.length >= 6) {
|
|
48
|
+
this.a = init[0];
|
|
49
|
+
this.b = init[1];
|
|
50
|
+
this.c = init[2];
|
|
51
|
+
this.d = init[3];
|
|
52
|
+
this.e = init[4];
|
|
53
|
+
this.f = init[5];
|
|
54
|
+
}
|
|
55
|
+
else {
|
|
56
|
+
this.a = this.d = 1;
|
|
57
|
+
this.b = this.c = this.e = this.f = 0;
|
|
58
|
+
}
|
|
59
|
+
}
|
|
60
|
+
// Map standard matrix properties for compatibility
|
|
61
|
+
get m11() { return this.a; }
|
|
62
|
+
get m12() { return this.b; }
|
|
63
|
+
get m21() { return this.c; }
|
|
64
|
+
get m22() { return this.d; }
|
|
65
|
+
get m41() { return this.e; }
|
|
66
|
+
get m42() { return this.f; }
|
|
67
|
+
};
|
|
68
|
+
}
|
|
69
|
+
}
|
|
@@ -7,10 +7,11 @@
|
|
|
7
7
|
* This approach prevents TypeScript from transpiling dynamic import()
|
|
8
8
|
* into require() when targeting CommonJS.
|
|
9
9
|
*/
|
|
10
|
+
import type * as FileTypeModule from 'file-type' with { 'resolution-mode': 'import' };
|
|
10
11
|
/**
|
|
11
12
|
* Specialized loader for file-type
|
|
12
13
|
*/
|
|
13
|
-
export declare function loadFileType(): Promise<typeof
|
|
14
|
+
export declare function loadFileType(): Promise<typeof FileTypeModule>;
|
|
14
15
|
/**
|
|
15
16
|
* Specialized loader for pdfjs-dist
|
|
16
17
|
*/
|
|
@@ -8,42 +8,10 @@
|
|
|
8
8
|
* This approach prevents TypeScript from transpiling dynamic import()
|
|
9
9
|
* into require() when targeting CommonJS.
|
|
10
10
|
*/
|
|
11
|
-
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
12
|
-
if (k2 === undefined) k2 = k;
|
|
13
|
-
var desc = Object.getOwnPropertyDescriptor(m, k);
|
|
14
|
-
if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
|
|
15
|
-
desc = { enumerable: true, get: function() { return m[k]; } };
|
|
16
|
-
}
|
|
17
|
-
Object.defineProperty(o, k2, desc);
|
|
18
|
-
}) : (function(o, m, k, k2) {
|
|
19
|
-
if (k2 === undefined) k2 = k;
|
|
20
|
-
o[k2] = m[k];
|
|
21
|
-
}));
|
|
22
|
-
var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
|
|
23
|
-
Object.defineProperty(o, "default", { enumerable: true, value: v });
|
|
24
|
-
}) : function(o, v) {
|
|
25
|
-
o["default"] = v;
|
|
26
|
-
});
|
|
27
|
-
var __importStar = (this && this.__importStar) || (function () {
|
|
28
|
-
var ownKeys = function(o) {
|
|
29
|
-
ownKeys = Object.getOwnPropertyNames || function (o) {
|
|
30
|
-
var ar = [];
|
|
31
|
-
for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
|
|
32
|
-
return ar;
|
|
33
|
-
};
|
|
34
|
-
return ownKeys(o);
|
|
35
|
-
};
|
|
36
|
-
return function (mod) {
|
|
37
|
-
if (mod && mod.__esModule) return mod;
|
|
38
|
-
var result = {};
|
|
39
|
-
if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
|
|
40
|
-
__setModuleDefault(result, mod);
|
|
41
|
-
return result;
|
|
42
|
-
};
|
|
43
|
-
})();
|
|
44
11
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
45
12
|
exports.loadFileType = loadFileType;
|
|
46
13
|
exports.loadPdfJs = loadPdfJs;
|
|
14
|
+
const envUtils_js_1 = require("./envUtils.js");
|
|
47
15
|
/**
|
|
48
16
|
* Dynamically loads an ESM module in a Node.js CJS context.
|
|
49
17
|
*
|
|
@@ -59,18 +27,20 @@ async function loadNodeEsmModule(specifier) {
|
|
|
59
27
|
* Specialized loader for file-type
|
|
60
28
|
*/
|
|
61
29
|
async function loadFileType() {
|
|
62
|
-
if (
|
|
63
|
-
// Node.js path
|
|
30
|
+
if (!envUtils_js_1.isBrowser) {
|
|
31
|
+
// Node.js path: Use dynamic import wrapper for CJS compatibility
|
|
64
32
|
return loadNodeEsmModule('file-type');
|
|
65
33
|
}
|
|
66
|
-
// Browser path:
|
|
67
|
-
return
|
|
34
|
+
// Browser path: standard dynamic import() is handled by bundlers (e.g. esbuild/Vite)
|
|
35
|
+
return import('file-type');
|
|
68
36
|
}
|
|
69
37
|
/**
|
|
70
38
|
* Specialized loader for pdfjs-dist
|
|
71
39
|
*/
|
|
72
40
|
async function loadPdfJs() {
|
|
73
|
-
if (
|
|
41
|
+
if (!envUtils_js_1.isBrowser) {
|
|
42
|
+
// Ensure DOMMatrix polyfill for Node.js 18 support
|
|
43
|
+
(0, envUtils_js_1.ensureDomMatrix)();
|
|
74
44
|
// Node.js environment: require legacy build for stability with ESM-only main
|
|
75
45
|
try {
|
|
76
46
|
return await loadNodeEsmModule('pdfjs-dist/legacy/build/pdf.mjs');
|
|
@@ -80,5 +50,5 @@ async function loadPdfJs() {
|
|
|
80
50
|
}
|
|
81
51
|
}
|
|
82
52
|
// Browser environment: esbuild handles standard static-looking dynamic import()
|
|
83
|
-
return
|
|
53
|
+
return import('pdfjs-dist');
|
|
84
54
|
}
|
package/dist/utils/ocrUtils.d.ts
CHANGED
|
@@ -4,8 +4,12 @@
|
|
|
4
4
|
* This module provides functions for extracting text from images using Tesseract.js.
|
|
5
5
|
* Used when `config.ocr` is enabled to extract text from embedded images in documents.
|
|
6
6
|
*
|
|
7
|
+
* Includes a worker pool via OcrSchedulerManager to improve performance when
|
|
8
|
+
* processing multiple images.
|
|
9
|
+
*
|
|
7
10
|
* @module ocrUtils
|
|
8
11
|
*/
|
|
12
|
+
import { OcrConfig } from '../types.js';
|
|
9
13
|
/**
|
|
10
14
|
* Performs Optical Character Recognition (OCR) on an image to extract text.
|
|
11
15
|
*
|
|
@@ -13,26 +17,26 @@
|
|
|
13
17
|
* This is useful for extracting text from screenshots, scanned documents,
|
|
14
18
|
* charts with labels, or any image containing text.
|
|
15
19
|
*
|
|
16
|
-
*
|
|
17
|
-
* and properly terminates the worker to free resources.
|
|
20
|
+
* This function uses a shared worker pool to minimize initialization overhead.
|
|
18
21
|
*
|
|
19
|
-
* @param
|
|
20
|
-
* @param
|
|
21
|
-
* Supports ISO 639-2/T three-letter codes: 'eng', 'spa', 'fra', 'deu', etc.
|
|
22
|
-
* Multiple languages can be combined with '+': 'eng+fra'
|
|
22
|
+
* @param image - The image data as a Buffer, file path, or Blob
|
|
23
|
+
* @param config - Optional configuration for language and custom worker paths
|
|
23
24
|
* @returns A promise that resolves to the recognized text as a string
|
|
24
25
|
* @throws {Error} If the image cannot be processed or Tesseract initialization fails
|
|
25
26
|
*
|
|
26
27
|
* @example
|
|
27
28
|
* ```typescript
|
|
28
29
|
* // Extract text from an English image
|
|
29
|
-
* const text = await performOcr(imageBuffer, 'eng');
|
|
30
|
-
* console.log(text); // "Annual Revenue: $1.2M"
|
|
31
|
-
*
|
|
32
|
-
* // Extract text from a multilingual image
|
|
33
|
-
* const text = await performOcr(imageBuffer, 'eng+spa');
|
|
30
|
+
* const text = await performOcr(imageBuffer, { language: 'eng' });
|
|
34
31
|
* ```
|
|
35
32
|
*
|
|
36
33
|
* @see https://github.com/naptha/tesseract.js for supported languages and options
|
|
37
34
|
*/
|
|
38
|
-
export declare const performOcr: (image: Buffer | string,
|
|
35
|
+
export declare const performOcr: (image: Buffer | string, config?: OcrConfig) => Promise<string>;
|
|
36
|
+
/**
|
|
37
|
+
* Terminates all OCR workers and cleans up resources.
|
|
38
|
+
*
|
|
39
|
+
* Should be called when the application is shutting down or OCR is no longer needed
|
|
40
|
+
* to prevent memory leaks and dangling worker processes.
|
|
41
|
+
*/
|
|
42
|
+
export declare const terminateOcr: () => Promise<void>;
|
package/dist/utils/ocrUtils.js
CHANGED
|
@@ -5,11 +5,176 @@
|
|
|
5
5
|
* This module provides functions for extracting text from images using Tesseract.js.
|
|
6
6
|
* Used when `config.ocr` is enabled to extract text from embedded images in documents.
|
|
7
7
|
*
|
|
8
|
+
* Includes a worker pool via OcrSchedulerManager to improve performance when
|
|
9
|
+
* processing multiple images.
|
|
10
|
+
*
|
|
8
11
|
* @module ocrUtils
|
|
9
12
|
*/
|
|
10
13
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
11
|
-
exports.performOcr = void 0;
|
|
12
|
-
|
|
14
|
+
exports.terminateOcr = exports.performOcr = void 0;
|
|
15
|
+
/**
|
|
16
|
+
* Manages a pool of Tesseract workers with "Smart Affinity".
|
|
17
|
+
*
|
|
18
|
+
* Instead of a simple scheduler, this manager allows workers to persist with
|
|
19
|
+
* a specific language affinity. If a new language is requested and the pool
|
|
20
|
+
* is at capacity, it re-initializes the Least Recently Used (LRU) idle worker
|
|
21
|
+
* rather than resetting the entire pool.
|
|
22
|
+
*
|
|
23
|
+
* Implements lazy loading of tesseract.js to ensure no background processes
|
|
24
|
+
* are spawned unless OCR is explicitly used.
|
|
25
|
+
*/
|
|
26
|
+
class OcrSchedulerManager {
|
|
27
|
+
static instance;
|
|
28
|
+
pool = [];
|
|
29
|
+
queue = [];
|
|
30
|
+
MAX_WORKERS = 4;
|
|
31
|
+
idleTimeout = 10000; // 10s default
|
|
32
|
+
timeoutId = null;
|
|
33
|
+
constructor() { }
|
|
34
|
+
/**
|
|
35
|
+
* Returns the singleton instance of the manager.
|
|
36
|
+
*/
|
|
37
|
+
static getInstance() {
|
|
38
|
+
if (!OcrSchedulerManager.instance) {
|
|
39
|
+
OcrSchedulerManager.instance = new OcrSchedulerManager();
|
|
40
|
+
}
|
|
41
|
+
return OcrSchedulerManager.instance;
|
|
42
|
+
}
|
|
43
|
+
/**
|
|
44
|
+
* Checks if the singleton instance has been initialized.
|
|
45
|
+
*/
|
|
46
|
+
static hasInstance() {
|
|
47
|
+
return !!OcrSchedulerManager.instance;
|
|
48
|
+
}
|
|
49
|
+
/**
|
|
50
|
+
* Resets the inactivity timer. If the timer reaches its duration,
|
|
51
|
+
* all workers are terminated automatically.
|
|
52
|
+
*/
|
|
53
|
+
resetIdleTimer() {
|
|
54
|
+
if (this.timeoutId) {
|
|
55
|
+
clearTimeout(this.timeoutId);
|
|
56
|
+
}
|
|
57
|
+
if (this.idleTimeout > 0) {
|
|
58
|
+
this.timeoutId = setTimeout(async () => {
|
|
59
|
+
await this.terminate();
|
|
60
|
+
}, this.idleTimeout);
|
|
61
|
+
}
|
|
62
|
+
}
|
|
63
|
+
/**
|
|
64
|
+
* Performs OCR on an image using the smart worker pool.
|
|
65
|
+
*
|
|
66
|
+
* @param image - Image data (Buffer, string path, or Blob)
|
|
67
|
+
* @param config - OCR configuration (language, custom paths)
|
|
68
|
+
* @returns Recognized text
|
|
69
|
+
*/
|
|
70
|
+
async recognize(image, config) {
|
|
71
|
+
return new Promise((resolve, reject) => {
|
|
72
|
+
// Update idle timeout if provided
|
|
73
|
+
if (config?.autoTerminateTimeout !== undefined) {
|
|
74
|
+
this.idleTimeout = config.autoTerminateTimeout;
|
|
75
|
+
}
|
|
76
|
+
// Reset the inactivity timer every time a new job is requested
|
|
77
|
+
this.resetIdleTimer();
|
|
78
|
+
// Add job to queue and trigger processing
|
|
79
|
+
this.queue.push({ image, config: config || {}, resolve, reject });
|
|
80
|
+
this.processQueue();
|
|
81
|
+
});
|
|
82
|
+
}
|
|
83
|
+
/**
|
|
84
|
+
* Attempts to process the next job in the queue using an available worker.
|
|
85
|
+
*/
|
|
86
|
+
async processQueue() {
|
|
87
|
+
if (this.queue.length === 0)
|
|
88
|
+
return;
|
|
89
|
+
const nextJob = this.queue[0];
|
|
90
|
+
const requestedLanguage = nextJob.config.language || 'eng';
|
|
91
|
+
// 1. Find an idle worker with the EXACT language affinity
|
|
92
|
+
let managed = this.pool.find(mw => !mw.isBusy && mw.language === requestedLanguage);
|
|
93
|
+
// 2. If not found and we have room, create a new worker
|
|
94
|
+
if (!managed && this.pool.length < this.MAX_WORKERS) {
|
|
95
|
+
try {
|
|
96
|
+
const { createWorker } = await import('tesseract.js');
|
|
97
|
+
const options = { logger: () => { } };
|
|
98
|
+
if (nextJob.config.workerPath)
|
|
99
|
+
options.workerPath = nextJob.config.workerPath;
|
|
100
|
+
if (nextJob.config.corePath)
|
|
101
|
+
options.corePath = nextJob.config.corePath;
|
|
102
|
+
if (nextJob.config.langPath)
|
|
103
|
+
options.langPath = nextJob.config.langPath;
|
|
104
|
+
const worker = await createWorker(requestedLanguage, 1, options);
|
|
105
|
+
managed = {
|
|
106
|
+
worker,
|
|
107
|
+
language: requestedLanguage,
|
|
108
|
+
lastUsed: Date.now(),
|
|
109
|
+
isBusy: false
|
|
110
|
+
};
|
|
111
|
+
this.pool.push(managed);
|
|
112
|
+
}
|
|
113
|
+
catch (err) {
|
|
114
|
+
const job = this.queue.shift();
|
|
115
|
+
job?.reject(err);
|
|
116
|
+
this.processQueue(); // Try next job
|
|
117
|
+
return;
|
|
118
|
+
}
|
|
119
|
+
}
|
|
120
|
+
// 3. If still not found and we are at capacity, find the LRU idle worker and re-initialize it
|
|
121
|
+
if (!managed) {
|
|
122
|
+
const idleWorkers = this.pool.filter(mw => !mw.isBusy);
|
|
123
|
+
if (idleWorkers.length > 0) {
|
|
124
|
+
// Find Least Recently Used idle worker
|
|
125
|
+
managed = idleWorkers.reduce((prev, curr) => (prev.lastUsed < curr.lastUsed ? prev : curr));
|
|
126
|
+
try {
|
|
127
|
+
// Smart Re-initialization (v5 API)
|
|
128
|
+
await managed.worker.reinitialize(requestedLanguage);
|
|
129
|
+
managed.language = requestedLanguage;
|
|
130
|
+
}
|
|
131
|
+
catch (err) {
|
|
132
|
+
// If reinitialization fails, we might need to recreate it, but for simplicity
|
|
133
|
+
// we'll just fail this job and try another worker next time.
|
|
134
|
+
const job = this.queue.shift();
|
|
135
|
+
job?.reject(err);
|
|
136
|
+
this.processQueue();
|
|
137
|
+
return;
|
|
138
|
+
}
|
|
139
|
+
}
|
|
140
|
+
}
|
|
141
|
+
// 4. If we have a worker ready, execute the job
|
|
142
|
+
if (managed) {
|
|
143
|
+
const job = this.queue.shift();
|
|
144
|
+
if (!job)
|
|
145
|
+
return;
|
|
146
|
+
managed.isBusy = true;
|
|
147
|
+
managed.lastUsed = Date.now();
|
|
148
|
+
try {
|
|
149
|
+
const { data: { text } } = await managed.worker.recognize(job.image);
|
|
150
|
+
job.resolve(text);
|
|
151
|
+
}
|
|
152
|
+
catch (err) {
|
|
153
|
+
job.reject(err);
|
|
154
|
+
}
|
|
155
|
+
finally {
|
|
156
|
+
managed.isBusy = false;
|
|
157
|
+
managed.lastUsed = Date.now();
|
|
158
|
+
// Check if there are more jobs waiting
|
|
159
|
+
this.processQueue();
|
|
160
|
+
}
|
|
161
|
+
}
|
|
162
|
+
// If no worker is available (all busy), the job stays in the queue
|
|
163
|
+
// and will be picked up when a worker finishes.
|
|
164
|
+
}
|
|
165
|
+
/**
|
|
166
|
+
* Terminates all workers in the pool and resets the state.
|
|
167
|
+
*/
|
|
168
|
+
async terminate() {
|
|
169
|
+
if (this.timeoutId) {
|
|
170
|
+
clearTimeout(this.timeoutId);
|
|
171
|
+
this.timeoutId = null;
|
|
172
|
+
}
|
|
173
|
+
const workersToTerminate = this.pool.map(mw => mw.worker.terminate());
|
|
174
|
+
await Promise.all(workersToTerminate);
|
|
175
|
+
this.pool = [];
|
|
176
|
+
}
|
|
177
|
+
}
|
|
13
178
|
/**
|
|
14
179
|
* Performs Optical Character Recognition (OCR) on an image to extract text.
|
|
15
180
|
*
|
|
@@ -17,45 +182,41 @@ const tesseract_js_1 = require("tesseract.js");
|
|
|
17
182
|
* This is useful for extracting text from screenshots, scanned documents,
|
|
18
183
|
* charts with labels, or any image containing text.
|
|
19
184
|
*
|
|
20
|
-
*
|
|
21
|
-
* and properly terminates the worker to free resources.
|
|
185
|
+
* This function uses a shared worker pool to minimize initialization overhead.
|
|
22
186
|
*
|
|
23
|
-
* @param
|
|
24
|
-
* @param
|
|
25
|
-
* Supports ISO 639-2/T three-letter codes: 'eng', 'spa', 'fra', 'deu', etc.
|
|
26
|
-
* Multiple languages can be combined with '+': 'eng+fra'
|
|
187
|
+
* @param image - The image data as a Buffer, file path, or Blob
|
|
188
|
+
* @param config - Optional configuration for language and custom worker paths
|
|
27
189
|
* @returns A promise that resolves to the recognized text as a string
|
|
28
190
|
* @throws {Error} If the image cannot be processed or Tesseract initialization fails
|
|
29
191
|
*
|
|
30
192
|
* @example
|
|
31
193
|
* ```typescript
|
|
32
194
|
* // Extract text from an English image
|
|
33
|
-
* const text = await performOcr(imageBuffer, 'eng');
|
|
34
|
-
* console.log(text); // "Annual Revenue: $1.2M"
|
|
35
|
-
*
|
|
36
|
-
* // Extract text from a multilingual image
|
|
37
|
-
* const text = await performOcr(imageBuffer, 'eng+spa');
|
|
195
|
+
* const text = await performOcr(imageBuffer, { language: 'eng' });
|
|
38
196
|
* ```
|
|
39
197
|
*
|
|
40
198
|
* @see https://github.com/naptha/tesseract.js for supported languages and options
|
|
41
199
|
*/
|
|
42
|
-
const performOcr = async (image,
|
|
43
|
-
//
|
|
44
|
-
// We pass 1 for OEM (LSTM) and a silent logger to suppress console output
|
|
45
|
-
const worker = await (0, tesseract_js_1.createWorker)(language, 1, {
|
|
46
|
-
logger: () => { }
|
|
47
|
-
});
|
|
48
|
-
// Step 2: Prepare image data
|
|
200
|
+
const performOcr = async (image, config) => {
|
|
201
|
+
// Prepare image data
|
|
49
202
|
let inputImage = image;
|
|
50
203
|
// In browser environment, convert Buffer to Blob for better compatibility
|
|
51
204
|
// @ts-ignore
|
|
52
205
|
if (typeof window !== 'undefined' && typeof Blob !== 'undefined' && Buffer.isBuffer(image)) {
|
|
53
206
|
inputImage = new Blob([image], { type: 'image/bmp' });
|
|
54
207
|
}
|
|
55
|
-
|
|
56
|
-
const ret = await worker.recognize(inputImage);
|
|
57
|
-
// Step 4: Terminate worker
|
|
58
|
-
await worker.terminate();
|
|
59
|
-
return ret.data.text;
|
|
208
|
+
return await OcrSchedulerManager.getInstance().recognize(inputImage, config);
|
|
60
209
|
};
|
|
61
210
|
exports.performOcr = performOcr;
|
|
211
|
+
/**
|
|
212
|
+
* Terminates all OCR workers and cleans up resources.
|
|
213
|
+
*
|
|
214
|
+
* Should be called when the application is shutting down or OCR is no longer needed
|
|
215
|
+
* to prevent memory leaks and dangling worker processes.
|
|
216
|
+
*/
|
|
217
|
+
const terminateOcr = async () => {
|
|
218
|
+
if (OcrSchedulerManager.hasInstance()) {
|
|
219
|
+
await OcrSchedulerManager.getInstance().terminate();
|
|
220
|
+
}
|
|
221
|
+
};
|
|
222
|
+
exports.terminateOcr = terminateOcr;
|