officeparser 6.0.7 → 6.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +136 -52
- package/dist/OfficeParser.d.ts +10 -1
- package/dist/OfficeParser.js +44 -56
- package/dist/cli.d.ts +20 -0
- package/dist/cli.js +117 -0
- package/dist/index.d.ts +4 -4
- package/dist/index.js +7 -59
- package/dist/index.mjs +18 -0
- package/dist/officeparser.browser.d.ts +133 -3
- package/dist/officeparser.browser.iife.js +115 -0
- package/dist/officeparser.browser.mjs +114 -0
- package/dist/parsers/ExcelParser.d.ts +1 -1
- package/dist/parsers/ExcelParser.js +76 -68
- package/dist/parsers/OpenOfficeParser.d.ts +1 -1
- package/dist/parsers/OpenOfficeParser.js +224 -159
- package/dist/parsers/PdfParser.d.ts +1 -1
- package/dist/parsers/PdfParser.js +98 -94
- package/dist/parsers/PowerPointParser.d.ts +1 -1
- package/dist/parsers/PowerPointParser.js +188 -179
- package/dist/parsers/RtfParser.d.ts +21 -1
- package/dist/parsers/RtfParser.js +117 -48
- package/dist/parsers/WordParser.d.ts +2 -1
- package/dist/parsers/WordParser.js +214 -123
- package/dist/sbom.cdx.json +1807 -0
- package/dist/types.d.ts +123 -3
- package/dist/utils/chartUtils.js +2 -0
- package/dist/utils/dateUtils.d.ts +17 -0
- package/dist/utils/dateUtils.js +69 -0
- package/dist/utils/envUtils.d.ts +24 -0
- package/dist/utils/envUtils.js +69 -0
- package/dist/utils/moduleLoader.d.ts +2 -1
- package/dist/utils/moduleLoader.js +9 -39
- package/dist/utils/ocrUtils.d.ts +16 -12
- package/dist/utils/ocrUtils.js +186 -25
- package/dist/utils/xmlUtils.d.ts +80 -9
- package/dist/utils/xmlUtils.js +236 -18
- package/dist/utils/zipUtils.js +6 -47
- package/package.json +31 -16
- package/dist/officeParserBundle@6.0.7.js +0 -154
- package/dist/officeparser.browser.js +0 -154
package/dist/types.d.ts
CHANGED
|
@@ -1,3 +1,40 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Configuration options for OCR.
|
|
3
|
+
*/
|
|
4
|
+
export interface OcrConfig {
|
|
5
|
+
/**
|
|
6
|
+
* Language for OCR.
|
|
7
|
+
* Default is 'eng'.
|
|
8
|
+
*
|
|
9
|
+
* You can provide multiple languages separated by a `+` sign (e.g., 'eng+fra' for English and French).
|
|
10
|
+
* The OCR engine will then attempt to recognize text in any of the specified languages.
|
|
11
|
+
*
|
|
12
|
+
* See the list of supported languages and their codes here:
|
|
13
|
+
* https://tesseract-ocr.github.io/tessdoc/Data-Files#data-files-for-version-400-november-29-2016
|
|
14
|
+
*/
|
|
15
|
+
language?: string;
|
|
16
|
+
/**
|
|
17
|
+
* Path to the Tesseract worker script.
|
|
18
|
+
* Primarily used for offline/air-gapped environments.
|
|
19
|
+
*/
|
|
20
|
+
workerPath?: string;
|
|
21
|
+
/**
|
|
22
|
+
* Path to the Tesseract core script.
|
|
23
|
+
* Primarily used for offline/air-gapped environments.
|
|
24
|
+
*/
|
|
25
|
+
corePath?: string;
|
|
26
|
+
/**
|
|
27
|
+
* Path for Tesseract language files (traineddata).
|
|
28
|
+
* Primarily used for offline/air-gapped environments.
|
|
29
|
+
*/
|
|
30
|
+
langPath?: string;
|
|
31
|
+
/**
|
|
32
|
+
* Timeout in milliseconds of inactivity before the OCR worker pool is automatically terminated.
|
|
33
|
+
* Set to 0 to disable auto-termination.
|
|
34
|
+
* Default is 10,000 (10 seconds).
|
|
35
|
+
*/
|
|
36
|
+
autoTerminateTimeout?: number;
|
|
37
|
+
}
|
|
1
38
|
/**
|
|
2
39
|
* Configuration options for the OfficeParser.
|
|
3
40
|
*/
|
|
@@ -40,6 +77,7 @@ export interface OfficeParserConfig {
|
|
|
40
77
|
*/
|
|
41
78
|
ocr?: boolean;
|
|
42
79
|
/**
|
|
80
|
+
* @deprecated Use `ocrConfig.language` instead.
|
|
43
81
|
* Language for OCR.
|
|
44
82
|
* Default is 'eng'.
|
|
45
83
|
*
|
|
@@ -50,14 +88,41 @@ export interface OfficeParserConfig {
|
|
|
50
88
|
* https://tesseract-ocr.github.io/tessdoc/Data-Files#data-files-for-version-400-november-29-2016
|
|
51
89
|
*/
|
|
52
90
|
ocrLanguage?: string;
|
|
91
|
+
/**
|
|
92
|
+
* Shared OCR configuration for worker pooling and offline support.
|
|
93
|
+
* If provided, `ocrLanguage` will be ignored in favor of `ocrConfig.language`.
|
|
94
|
+
*/
|
|
95
|
+
ocrConfig?: OcrConfig;
|
|
96
|
+
/**
|
|
97
|
+
* Flag to serialize raw content (XML) as clean, formatted strings.
|
|
98
|
+
* Only relevant when `includeRawContent` is true.
|
|
99
|
+
* Default is true.
|
|
100
|
+
*
|
|
101
|
+
* If false, the parser will attempt to extract the original raw substring from the
|
|
102
|
+
* source document instead of re-serializing the DOM node.
|
|
103
|
+
*/
|
|
104
|
+
serializeRawContent?: boolean;
|
|
105
|
+
/**
|
|
106
|
+
* Flag to preserve original XML whitespace and line endings when serializing.
|
|
107
|
+
* Only relevant when `includeRawContent` is true and `serializeRawContent` is true.
|
|
108
|
+
* Default is false.
|
|
109
|
+
*/
|
|
110
|
+
preserveXmlWhitespace?: boolean;
|
|
53
111
|
/**
|
|
54
112
|
* The URL/path to the PDF.js worker script.
|
|
55
113
|
*
|
|
56
114
|
* **Mandatory** when using PDF parsing in browser environments to avoid worker configuration errors.
|
|
57
|
-
* If not provided, it defaults to `https://unpkg.com/pdfjs-dist@5.
|
|
115
|
+
* If not provided, it defaults to `https://unpkg.com/pdfjs-dist@5.6.205/build/pdf.worker.min.mjs`.
|
|
58
116
|
* You can override this with your own local path or a different CDN link.
|
|
59
117
|
*/
|
|
60
118
|
pdfWorkerSrc?: string;
|
|
119
|
+
/**
|
|
120
|
+
* Flag to include break nodes in the AST.
|
|
121
|
+
* This is currently only supported for Word documents. (w:br nodes)
|
|
122
|
+
*
|
|
123
|
+
* Default is false
|
|
124
|
+
*/
|
|
125
|
+
includeBreakNodes?: boolean;
|
|
61
126
|
}
|
|
62
127
|
/**
|
|
63
128
|
* Supported file types for parsing.
|
|
@@ -66,7 +131,7 @@ export type SupportedFileType = 'docx' | 'pptx' | 'xlsx' | 'odt' | 'odp' | 'ods'
|
|
|
66
131
|
/**
|
|
67
132
|
* Types of content nodes in the AST.
|
|
68
133
|
*/
|
|
69
|
-
export type OfficeContentNodeType = 'paragraph' | 'heading' | 'table' | 'list' | 'text' | 'image' | 'chart' | 'drawing' | 'slide' | 'note' | 'sheet' | 'row' | 'cell' | 'page';
|
|
134
|
+
export type OfficeContentNodeType = 'paragraph' | 'heading' | 'table' | 'list' | 'text' | 'image' | 'chart' | 'drawing' | 'slide' | 'note' | 'sheet' | 'row' | 'cell' | 'page' | 'break';
|
|
70
135
|
/**
|
|
71
136
|
* Supported MIME types for attachments.
|
|
72
137
|
*/
|
|
@@ -169,6 +234,20 @@ export interface SheetMetadata {
|
|
|
169
234
|
/** The style of the sheet. */
|
|
170
235
|
style?: string;
|
|
171
236
|
}
|
|
237
|
+
/**
|
|
238
|
+
* Detailed indentation information for paragraphs and headings.
|
|
239
|
+
* Values are typically in twentieths of a point (twips) in OOXML.
|
|
240
|
+
*/
|
|
241
|
+
export interface IndentationMetadata {
|
|
242
|
+
/** Left indentation. */
|
|
243
|
+
left?: number;
|
|
244
|
+
/** Right indentation. */
|
|
245
|
+
right?: number;
|
|
246
|
+
/** First line indentation. */
|
|
247
|
+
firstLine?: number;
|
|
248
|
+
/** Hanging indentation. */
|
|
249
|
+
hanging?: number;
|
|
250
|
+
}
|
|
172
251
|
/**
|
|
173
252
|
* Metadata for a heading.
|
|
174
253
|
*/
|
|
@@ -179,6 +258,8 @@ export interface HeadingMetadata {
|
|
|
179
258
|
alignment?: 'left' | 'center' | 'right' | 'justify';
|
|
180
259
|
/** The style of the heading. */
|
|
181
260
|
style?: string;
|
|
261
|
+
/** Detailed indentation information. */
|
|
262
|
+
paragraphIndentation?: IndentationMetadata;
|
|
182
263
|
}
|
|
183
264
|
/**
|
|
184
265
|
* Metadata for a paragraph.
|
|
@@ -188,6 +269,8 @@ export interface ParagraphMetadata {
|
|
|
188
269
|
alignment?: 'left' | 'center' | 'right' | 'justify';
|
|
189
270
|
/** The style of the paragraph. */
|
|
190
271
|
style?: string;
|
|
272
|
+
/** Detailed indentation information. */
|
|
273
|
+
paragraphIndentation?: IndentationMetadata;
|
|
191
274
|
}
|
|
192
275
|
/**
|
|
193
276
|
* Metadata for a list item.
|
|
@@ -203,6 +286,8 @@ export interface ListMetadata {
|
|
|
203
286
|
* @example 0 for top-level items, 1 for first nested level
|
|
204
287
|
*/
|
|
205
288
|
indentation: number;
|
|
289
|
+
/** Detailed indentation information. */
|
|
290
|
+
paragraphIndentation?: IndentationMetadata;
|
|
206
291
|
/**
|
|
207
292
|
* Text alignment of the list item.
|
|
208
293
|
* @example 'left', 'center', 'right', 'justify'
|
|
@@ -329,10 +414,35 @@ export interface NoteMetadata {
|
|
|
329
414
|
*/
|
|
330
415
|
noteId?: string;
|
|
331
416
|
}
|
|
417
|
+
/**
|
|
418
|
+
* Metadata for break nodes.
|
|
419
|
+
* Used in DOCX files to track line and page breaks.
|
|
420
|
+
*/
|
|
421
|
+
export interface BreakMetadata {
|
|
422
|
+
/**
|
|
423
|
+
* Type of break. The break type determines the next location where
|
|
424
|
+
* text shall be placed.
|
|
425
|
+
* - 'column': The next text will be placed in the next column.
|
|
426
|
+
* - 'page': The next text will be placed on the next page.
|
|
427
|
+
* - 'lastRenderedPage': The editing application has inserted a soft break on the last save.
|
|
428
|
+
* - 'textWrapping' (default, assumed when not specified): The next text will be placed on the next line.
|
|
429
|
+
* - 'carriageReturn': An explicit carriage return (w:cr) equivalent to a hard line break.
|
|
430
|
+
*/
|
|
431
|
+
breakType: 'column' | 'page' | 'lastRenderedPage' | 'textWrapping' | 'carriageReturn';
|
|
432
|
+
/**
|
|
433
|
+
* Specifies the location which shall be used as the next available line when breakType
|
|
434
|
+
* has a value of 'textWrapping'. Should be ignored for other break types.
|
|
435
|
+
* - 'all': text wrapping break shall advance the text to the next line which spans the full width of the line
|
|
436
|
+
* - 'left': text wrapping break shall restart in next text region unblocked on the left
|
|
437
|
+
* - 'none': text wrapping break shall advance the text to the next line regardless of any floating objects
|
|
438
|
+
* - 'right': text wrapping break shall restart in next text region unblocked on the right
|
|
439
|
+
*/
|
|
440
|
+
clear?: 'all' | 'left' | 'none' | 'right';
|
|
441
|
+
}
|
|
332
442
|
/**
|
|
333
443
|
* Union type for content metadata.
|
|
334
444
|
*/
|
|
335
|
-
export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata | ListMetadata | CellMetadata | ImageMetadata | ChartMetadata | PageMetadata | ParagraphMetadata | TextMetadata | NoteMetadata | undefined;
|
|
445
|
+
export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata | ListMetadata | CellMetadata | ImageMetadata | ChartMetadata | PageMetadata | ParagraphMetadata | TextMetadata | NoteMetadata | BreakMetadata | undefined;
|
|
336
446
|
/**
|
|
337
447
|
* Represents a node in the document content tree.
|
|
338
448
|
* This is the core building block of the parsed document structure.
|
|
@@ -540,6 +650,16 @@ export interface OfficeMetadata {
|
|
|
540
650
|
formatting?: Partial<TextFormatting>;
|
|
541
651
|
/** Style map for styles in the document. */
|
|
542
652
|
styleMap?: Record<string, Partial<TextFormatting>>;
|
|
653
|
+
/**
|
|
654
|
+
* User-defined custom properties embedded in the document.
|
|
655
|
+
* Sources by format:
|
|
656
|
+
* - DOCX/XLSX/PPTX: `docProps/custom.xml` (Office custom document properties)
|
|
657
|
+
* - ODT/ODP/ODS: `meta:user-defined` elements in `meta.xml`
|
|
658
|
+
* - PDF: non-standard entries in the PDF Info dictionary
|
|
659
|
+
* RTF does not support custom properties; the `\info` group is not extracted.
|
|
660
|
+
* Values are typed as string, number, boolean, or Date where the source format provides type information.
|
|
661
|
+
*/
|
|
662
|
+
customProperties?: Record<string, string | number | boolean | Date>;
|
|
543
663
|
}
|
|
544
664
|
/**
|
|
545
665
|
* The Abstract Syntax Tree (AST) returned by the parser.
|
package/dist/utils/chartUtils.js
CHANGED
|
@@ -43,6 +43,8 @@ const extractOpenXmlChartData = (xmlBuffer) => {
|
|
|
43
43
|
const xml = xmlBuffer.toString("utf8");
|
|
44
44
|
const dom = (0, xmlUtils_1.parseXmlString)(xml);
|
|
45
45
|
const root = dom.documentElement;
|
|
46
|
+
if (!root)
|
|
47
|
+
return { title: undefined, xAxisTitle: undefined, yAxisTitle: undefined, dataSets: [], labels: [], rawTexts: [] };
|
|
46
48
|
const title = extractOpenXmlRichText(root, "c:title");
|
|
47
49
|
// Axis titles
|
|
48
50
|
let xAxisTitle = undefined;
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Date Parsing Utilities
|
|
3
|
+
*
|
|
4
|
+
* Provides robust functions for parsing date strings from various office formats.
|
|
5
|
+
* Handles standard ISO dates, PDF-specific date formats, and malformed strings.
|
|
6
|
+
*
|
|
7
|
+
* @module dateUtils
|
|
8
|
+
*/
|
|
9
|
+
/**
|
|
10
|
+
* Parses a date string into a Date object.
|
|
11
|
+
* Handles standard ISO formats and falls back to native parsing.
|
|
12
|
+
* Returns undefined instead of "Invalid Date" if parsing fails.
|
|
13
|
+
*
|
|
14
|
+
* @param dateString - The date string to parse
|
|
15
|
+
* @returns Parsed Date object or undefined if parsing fails
|
|
16
|
+
*/
|
|
17
|
+
export declare function parseOfficeDate(dateString: string | undefined): Date | undefined;
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* Date Parsing Utilities
|
|
4
|
+
*
|
|
5
|
+
* Provides robust functions for parsing date strings from various office formats.
|
|
6
|
+
* Handles standard ISO dates, PDF-specific date formats, and malformed strings.
|
|
7
|
+
*
|
|
8
|
+
* @module dateUtils
|
|
9
|
+
*/
|
|
10
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
11
|
+
exports.parseOfficeDate = parseOfficeDate;
|
|
12
|
+
/**
|
|
13
|
+
* Parses a date string into a Date object.
|
|
14
|
+
* Handles standard ISO formats and falls back to native parsing.
|
|
15
|
+
* Returns undefined instead of "Invalid Date" if parsing fails.
|
|
16
|
+
*
|
|
17
|
+
* @param dateString - The date string to parse
|
|
18
|
+
* @returns Parsed Date object or undefined if parsing fails
|
|
19
|
+
*/
|
|
20
|
+
function parseOfficeDate(dateString) {
|
|
21
|
+
if (!dateString)
|
|
22
|
+
return undefined;
|
|
23
|
+
try {
|
|
24
|
+
// PDF-specific format detection: D:YYYYMMDDHHmmSSOHH'mm'
|
|
25
|
+
if (dateString.startsWith('D:')) {
|
|
26
|
+
return parsePdfDate(dateString);
|
|
27
|
+
}
|
|
28
|
+
const date = new Date(dateString);
|
|
29
|
+
return isNaN(date.getTime()) ? undefined : date;
|
|
30
|
+
}
|
|
31
|
+
catch {
|
|
32
|
+
return undefined;
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
/**
|
|
36
|
+
* Internal helper for PDF-specific date format: D:YYYYMMDDHHmmSSOHH'mm'
|
|
37
|
+
* @param dateString - The PDF date string
|
|
38
|
+
*/
|
|
39
|
+
function parsePdfDate(dateString) {
|
|
40
|
+
try {
|
|
41
|
+
// Remove "D:" prefix
|
|
42
|
+
let str = dateString.slice(2);
|
|
43
|
+
// Extract components: YYYYMMDDHHmmSS
|
|
44
|
+
const year = parseInt(str.slice(0, 4), 10);
|
|
45
|
+
const month = parseInt(str.slice(4, 6), 10) - 1; // 0-indexed
|
|
46
|
+
const day = parseInt(str.slice(6, 8), 10) || 1;
|
|
47
|
+
const hour = parseInt(str.slice(8, 10), 10) || 0;
|
|
48
|
+
const minute = parseInt(str.slice(10, 12), 10) || 0;
|
|
49
|
+
const second = parseInt(str.slice(12, 14), 10) || 0;
|
|
50
|
+
// Handle timezone if present
|
|
51
|
+
const tzMatch = str.slice(14).match(/([+-Z])(\d{2})'?(\d{2})?'?/);
|
|
52
|
+
if (tzMatch) {
|
|
53
|
+
if (tzMatch[1] === 'Z') {
|
|
54
|
+
return new Date(Date.UTC(year, month, day, hour, minute, second));
|
|
55
|
+
}
|
|
56
|
+
const tzSign = tzMatch[1] === '-' ? -1 : 1;
|
|
57
|
+
const tzHours = parseInt(tzMatch[2], 10) || 0;
|
|
58
|
+
const tzMinutes = parseInt(tzMatch[3], 10) || 0;
|
|
59
|
+
const offset = tzSign * (tzHours * 60 + tzMinutes);
|
|
60
|
+
// Create date in UTC and adjust for timezone
|
|
61
|
+
const utc = Date.UTC(year, month, day, hour, minute, second);
|
|
62
|
+
return new Date(utc - offset * 60000);
|
|
63
|
+
}
|
|
64
|
+
return new Date(year, month, day, hour, minute, second);
|
|
65
|
+
}
|
|
66
|
+
catch {
|
|
67
|
+
return undefined;
|
|
68
|
+
}
|
|
69
|
+
}
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Environment detection and safe utility wrappers.
|
|
3
|
+
*/
|
|
4
|
+
/**
|
|
5
|
+
* Detect if we are running in a browser environment.
|
|
6
|
+
*/
|
|
7
|
+
export declare const isBrowser: boolean;
|
|
8
|
+
/**
|
|
9
|
+
* Supported Node.js-only features that require explicit guarding for browser compatibility.
|
|
10
|
+
*/
|
|
11
|
+
export type NodeFeature = 'fs' | 'path-parsing' | 'pdf-worker-auto-resolution';
|
|
12
|
+
/**
|
|
13
|
+
* Throws an error if attempted to use Node.js-specific features in the browser.
|
|
14
|
+
*
|
|
15
|
+
* @param feature - The Node.js feature being accessed
|
|
16
|
+
* @throws {Error} Clear error message directing browser users to use Buffers
|
|
17
|
+
*/
|
|
18
|
+
export declare function assertNode(feature: NodeFeature): void;
|
|
19
|
+
/**
|
|
20
|
+
* Polyfills DOMMatrix if not available globally (required for Node.js < 20).
|
|
21
|
+
* This shim provides enough properties for pdfjs-dist 5.x to calculate
|
|
22
|
+
* text coordinates and transformations.
|
|
23
|
+
*/
|
|
24
|
+
export declare function ensureDomMatrix(): void;
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* Environment detection and safe utility wrappers.
|
|
4
|
+
*/
|
|
5
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
6
|
+
exports.isBrowser = void 0;
|
|
7
|
+
exports.assertNode = assertNode;
|
|
8
|
+
exports.ensureDomMatrix = ensureDomMatrix;
|
|
9
|
+
/**
|
|
10
|
+
* Detect if we are running in a browser environment.
|
|
11
|
+
*/
|
|
12
|
+
exports.isBrowser = typeof window !== 'undefined' && typeof window.document !== 'undefined';
|
|
13
|
+
/**
|
|
14
|
+
* Human-readable descriptions for Node-only features.
|
|
15
|
+
*/
|
|
16
|
+
const readableFeatures = {
|
|
17
|
+
'fs': 'direct file system access',
|
|
18
|
+
'path-parsing': 'parsing from file path string',
|
|
19
|
+
'pdf-worker-auto-resolution': 'automatic PDF worker resolution from node_modules'
|
|
20
|
+
};
|
|
21
|
+
/**
|
|
22
|
+
* Throws an error if attempted to use Node.js-specific features in the browser.
|
|
23
|
+
*
|
|
24
|
+
* @param feature - The Node.js feature being accessed
|
|
25
|
+
* @throws {Error} Clear error message directing browser users to use Buffers
|
|
26
|
+
*/
|
|
27
|
+
function assertNode(feature) {
|
|
28
|
+
if (exports.isBrowser) {
|
|
29
|
+
throw new Error(`officeparser: '${readableFeatures[feature]}' is not supported in the browser. Browser users must pass file content as Buffer or ArrayBuffer directly.`);
|
|
30
|
+
}
|
|
31
|
+
}
|
|
32
|
+
/**
|
|
33
|
+
* Polyfills DOMMatrix if not available globally (required for Node.js < 20).
|
|
34
|
+
* This shim provides enough properties for pdfjs-dist 5.x to calculate
|
|
35
|
+
* text coordinates and transformations.
|
|
36
|
+
*/
|
|
37
|
+
function ensureDomMatrix() {
|
|
38
|
+
if (typeof global !== 'undefined' && !global.DOMMatrix) {
|
|
39
|
+
global.DOMMatrix = class DOMMatrix {
|
|
40
|
+
a;
|
|
41
|
+
b;
|
|
42
|
+
c;
|
|
43
|
+
d;
|
|
44
|
+
e;
|
|
45
|
+
f;
|
|
46
|
+
constructor(init) {
|
|
47
|
+
if (Array.isArray(init) && init.length >= 6) {
|
|
48
|
+
this.a = init[0];
|
|
49
|
+
this.b = init[1];
|
|
50
|
+
this.c = init[2];
|
|
51
|
+
this.d = init[3];
|
|
52
|
+
this.e = init[4];
|
|
53
|
+
this.f = init[5];
|
|
54
|
+
}
|
|
55
|
+
else {
|
|
56
|
+
this.a = this.d = 1;
|
|
57
|
+
this.b = this.c = this.e = this.f = 0;
|
|
58
|
+
}
|
|
59
|
+
}
|
|
60
|
+
// Map standard matrix properties for compatibility
|
|
61
|
+
get m11() { return this.a; }
|
|
62
|
+
get m12() { return this.b; }
|
|
63
|
+
get m21() { return this.c; }
|
|
64
|
+
get m22() { return this.d; }
|
|
65
|
+
get m41() { return this.e; }
|
|
66
|
+
get m42() { return this.f; }
|
|
67
|
+
};
|
|
68
|
+
}
|
|
69
|
+
}
|
|
@@ -7,10 +7,11 @@
|
|
|
7
7
|
* This approach prevents TypeScript from transpiling dynamic import()
|
|
8
8
|
* into require() when targeting CommonJS.
|
|
9
9
|
*/
|
|
10
|
+
import type * as FileTypeModule from 'file-type' with { 'resolution-mode': 'import' };
|
|
10
11
|
/**
|
|
11
12
|
* Specialized loader for file-type
|
|
12
13
|
*/
|
|
13
|
-
export declare function loadFileType(): Promise<typeof
|
|
14
|
+
export declare function loadFileType(): Promise<typeof FileTypeModule>;
|
|
14
15
|
/**
|
|
15
16
|
* Specialized loader for pdfjs-dist
|
|
16
17
|
*/
|
|
@@ -8,42 +8,10 @@
|
|
|
8
8
|
* This approach prevents TypeScript from transpiling dynamic import()
|
|
9
9
|
* into require() when targeting CommonJS.
|
|
10
10
|
*/
|
|
11
|
-
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
12
|
-
if (k2 === undefined) k2 = k;
|
|
13
|
-
var desc = Object.getOwnPropertyDescriptor(m, k);
|
|
14
|
-
if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
|
|
15
|
-
desc = { enumerable: true, get: function() { return m[k]; } };
|
|
16
|
-
}
|
|
17
|
-
Object.defineProperty(o, k2, desc);
|
|
18
|
-
}) : (function(o, m, k, k2) {
|
|
19
|
-
if (k2 === undefined) k2 = k;
|
|
20
|
-
o[k2] = m[k];
|
|
21
|
-
}));
|
|
22
|
-
var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
|
|
23
|
-
Object.defineProperty(o, "default", { enumerable: true, value: v });
|
|
24
|
-
}) : function(o, v) {
|
|
25
|
-
o["default"] = v;
|
|
26
|
-
});
|
|
27
|
-
var __importStar = (this && this.__importStar) || (function () {
|
|
28
|
-
var ownKeys = function(o) {
|
|
29
|
-
ownKeys = Object.getOwnPropertyNames || function (o) {
|
|
30
|
-
var ar = [];
|
|
31
|
-
for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
|
|
32
|
-
return ar;
|
|
33
|
-
};
|
|
34
|
-
return ownKeys(o);
|
|
35
|
-
};
|
|
36
|
-
return function (mod) {
|
|
37
|
-
if (mod && mod.__esModule) return mod;
|
|
38
|
-
var result = {};
|
|
39
|
-
if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
|
|
40
|
-
__setModuleDefault(result, mod);
|
|
41
|
-
return result;
|
|
42
|
-
};
|
|
43
|
-
})();
|
|
44
11
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
45
12
|
exports.loadFileType = loadFileType;
|
|
46
13
|
exports.loadPdfJs = loadPdfJs;
|
|
14
|
+
const envUtils_js_1 = require("./envUtils.js");
|
|
47
15
|
/**
|
|
48
16
|
* Dynamically loads an ESM module in a Node.js CJS context.
|
|
49
17
|
*
|
|
@@ -59,18 +27,20 @@ async function loadNodeEsmModule(specifier) {
|
|
|
59
27
|
* Specialized loader for file-type
|
|
60
28
|
*/
|
|
61
29
|
async function loadFileType() {
|
|
62
|
-
if (
|
|
63
|
-
// Node.js path
|
|
30
|
+
if (!envUtils_js_1.isBrowser) {
|
|
31
|
+
// Node.js path: Use dynamic import wrapper for CJS compatibility
|
|
64
32
|
return loadNodeEsmModule('file-type');
|
|
65
33
|
}
|
|
66
|
-
// Browser path:
|
|
67
|
-
return
|
|
34
|
+
// Browser path: standard dynamic import() is handled by bundlers (e.g. esbuild/Vite)
|
|
35
|
+
return import('file-type');
|
|
68
36
|
}
|
|
69
37
|
/**
|
|
70
38
|
* Specialized loader for pdfjs-dist
|
|
71
39
|
*/
|
|
72
40
|
async function loadPdfJs() {
|
|
73
|
-
if (
|
|
41
|
+
if (!envUtils_js_1.isBrowser) {
|
|
42
|
+
// Ensure DOMMatrix polyfill for Node.js 18 support
|
|
43
|
+
(0, envUtils_js_1.ensureDomMatrix)();
|
|
74
44
|
// Node.js environment: require legacy build for stability with ESM-only main
|
|
75
45
|
try {
|
|
76
46
|
return await loadNodeEsmModule('pdfjs-dist/legacy/build/pdf.mjs');
|
|
@@ -80,5 +50,5 @@ async function loadPdfJs() {
|
|
|
80
50
|
}
|
|
81
51
|
}
|
|
82
52
|
// Browser environment: esbuild handles standard static-looking dynamic import()
|
|
83
|
-
return
|
|
53
|
+
return import('pdfjs-dist');
|
|
84
54
|
}
|
package/dist/utils/ocrUtils.d.ts
CHANGED
|
@@ -4,8 +4,12 @@
|
|
|
4
4
|
* This module provides functions for extracting text from images using Tesseract.js.
|
|
5
5
|
* Used when `config.ocr` is enabled to extract text from embedded images in documents.
|
|
6
6
|
*
|
|
7
|
+
* Includes a worker pool via OcrSchedulerManager to improve performance when
|
|
8
|
+
* processing multiple images.
|
|
9
|
+
*
|
|
7
10
|
* @module ocrUtils
|
|
8
11
|
*/
|
|
12
|
+
import { OcrConfig } from '../types.js';
|
|
9
13
|
/**
|
|
10
14
|
* Performs Optical Character Recognition (OCR) on an image to extract text.
|
|
11
15
|
*
|
|
@@ -13,26 +17,26 @@
|
|
|
13
17
|
* This is useful for extracting text from screenshots, scanned documents,
|
|
14
18
|
* charts with labels, or any image containing text.
|
|
15
19
|
*
|
|
16
|
-
*
|
|
17
|
-
* and properly terminates the worker to free resources.
|
|
20
|
+
* This function uses a shared worker pool to minimize initialization overhead.
|
|
18
21
|
*
|
|
19
|
-
* @param
|
|
20
|
-
* @param
|
|
21
|
-
* Supports ISO 639-2/T three-letter codes: 'eng', 'spa', 'fra', 'deu', etc.
|
|
22
|
-
* Multiple languages can be combined with '+': 'eng+fra'
|
|
22
|
+
* @param image - The image data as a Buffer, file path, or Blob
|
|
23
|
+
* @param config - Optional configuration for language and custom worker paths
|
|
23
24
|
* @returns A promise that resolves to the recognized text as a string
|
|
24
25
|
* @throws {Error} If the image cannot be processed or Tesseract initialization fails
|
|
25
26
|
*
|
|
26
27
|
* @example
|
|
27
28
|
* ```typescript
|
|
28
29
|
* // Extract text from an English image
|
|
29
|
-
* const text = await performOcr(imageBuffer, 'eng');
|
|
30
|
-
* console.log(text); // "Annual Revenue: $1.2M"
|
|
31
|
-
*
|
|
32
|
-
* // Extract text from a multilingual image
|
|
33
|
-
* const text = await performOcr(imageBuffer, 'eng+spa');
|
|
30
|
+
* const text = await performOcr(imageBuffer, { language: 'eng' });
|
|
34
31
|
* ```
|
|
35
32
|
*
|
|
36
33
|
* @see https://github.com/naptha/tesseract.js for supported languages and options
|
|
37
34
|
*/
|
|
38
|
-
export declare const performOcr: (image: Buffer | string,
|
|
35
|
+
export declare const performOcr: (image: Buffer | string, config?: OcrConfig) => Promise<string>;
|
|
36
|
+
/**
|
|
37
|
+
* Terminates all OCR workers and cleans up resources.
|
|
38
|
+
*
|
|
39
|
+
* Should be called when the application is shutting down or OCR is no longer needed
|
|
40
|
+
* to prevent memory leaks and dangling worker processes.
|
|
41
|
+
*/
|
|
42
|
+
export declare const terminateOcr: () => Promise<void>;
|