officeparser 6.1.1 → 7.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +301 -26
- package/dist/OfficeConverter.d.ts +46 -0
- package/dist/OfficeConverter.js +72 -0
- package/dist/OfficeGenerator.d.ts +19 -0
- package/dist/OfficeGenerator.js +48 -0
- package/dist/OfficeParser.d.ts +6 -0
- package/dist/OfficeParser.js +74 -31
- package/dist/cli.d.ts +3 -1
- package/dist/cli.js +106 -22
- package/dist/defaults.d.ts +41 -0
- package/dist/defaults.js +172 -0
- package/dist/generators/BaseGenerator.d.ts +58 -0
- package/dist/generators/BaseGenerator.js +107 -0
- package/dist/generators/ChunkingGenerator.d.ts +81 -0
- package/dist/generators/ChunkingGenerator.js +683 -0
- package/dist/generators/CsvGenerator.d.ts +30 -0
- package/dist/generators/CsvGenerator.js +233 -0
- package/dist/generators/HtmlGenerator.d.ts +37 -0
- package/dist/generators/HtmlGenerator.js +1013 -0
- package/dist/generators/MarkdownGenerator.d.ts +59 -0
- package/dist/generators/MarkdownGenerator.js +481 -0
- package/dist/generators/PdfGenerator.d.ts +22 -0
- package/dist/generators/PdfGenerator.js +118 -0
- package/dist/generators/RtfGenerator.d.ts +15 -0
- package/dist/generators/RtfGenerator.js +208 -0
- package/dist/generators/TextGenerator.d.ts +13 -0
- package/dist/generators/TextGenerator.js +108 -0
- package/dist/index.d.ts +11 -3
- package/dist/index.js +17 -2
- package/dist/index.mjs +2 -2
- package/dist/officeparser.browser.d.ts +828 -5
- package/dist/officeparser.browser.iife.js +703 -52
- package/dist/officeparser.browser.mjs +703 -52
- package/dist/parsers/CsvParser.d.ts +9 -0
- package/dist/parsers/CsvParser.js +110 -0
- package/dist/parsers/ExcelParser.d.ts +2 -2
- package/dist/parsers/ExcelParser.js +145 -114
- package/dist/parsers/HtmlParser.d.ts +2 -0
- package/dist/parsers/HtmlParser.js +539 -0
- package/dist/parsers/MarkdownParser.d.ts +2 -0
- package/dist/parsers/MarkdownParser.js +360 -0
- package/dist/parsers/OpenOfficeParser.d.ts +2 -2
- package/dist/parsers/OpenOfficeParser.js +140 -79
- package/dist/parsers/PdfParser.d.ts +2 -2
- package/dist/parsers/PdfParser.js +52 -49
- package/dist/parsers/PowerPointParser.d.ts +2 -2
- package/dist/parsers/PowerPointParser.js +20 -23
- package/dist/parsers/RtfParser.d.ts +2 -2
- package/dist/parsers/RtfParser.js +1291 -1240
- package/dist/parsers/WordParser.d.ts +2 -2
- package/dist/parsers/WordParser.js +232 -97
- package/dist/sbom.cdx.json +99 -99
- package/dist/types.d.ts +783 -5
- package/dist/types.js +73 -0
- package/dist/utils/astUtils.d.ts +16 -0
- package/dist/utils/astUtils.js +32 -0
- package/dist/utils/configUtils.d.ts +26 -0
- package/dist/utils/configUtils.js +140 -0
- package/dist/utils/envUtils.d.ts +8 -3
- package/dist/utils/envUtils.js +117 -34
- package/dist/utils/errorUtils.d.ts +17 -29
- package/dist/utils/errorUtils.js +110 -52
- package/dist/utils/moduleLoader.js +19 -11
- package/dist/utils/ocrUtils.js +2 -1
- package/dist/utils/sheetUtils.d.ts +7 -0
- package/dist/utils/sheetUtils.js +35 -0
- package/dist/utils/styleMapper.d.ts +36 -0
- package/dist/utils/styleMapper.js +224 -0
- package/dist/utils/xmlUtils.d.ts +0 -8
- package/dist/utils/xmlUtils.js +2 -1
- package/package.json +26 -7
package/dist/utils/errorUtils.js
CHANGED
|
@@ -7,46 +7,63 @@
|
|
|
7
7
|
* consistent error reporting across all parsers and the main entry point.
|
|
8
8
|
*/
|
|
9
9
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
10
|
-
exports.logWarning = exports.getWrappedError = exports.getOfficeError = exports.
|
|
10
|
+
exports.logWarning = exports.getWrappedError = exports.getOfficeError = exports.getWarningMessage = void 0;
|
|
11
|
+
const types_js_1 = require("../types.js");
|
|
11
12
|
/** Error header prefix for all error messages */
|
|
12
13
|
const ERRORHEADER = "[OfficeParser]: ";
|
|
13
|
-
/**
|
|
14
|
-
* Standard error types for OfficeParser.
|
|
15
|
-
* Use these to identify the kind of error being reported.
|
|
16
|
-
*/
|
|
17
|
-
var OfficeErrorType;
|
|
18
|
-
(function (OfficeErrorType) {
|
|
19
|
-
/** Unsupported file extension */
|
|
20
|
-
OfficeErrorType["EXTENSION_UNSUPPORTED"] = "EXTENSION_UNSUPPORTED";
|
|
21
|
-
/** File appears to be corrupted or malformed */
|
|
22
|
-
OfficeErrorType["FILE_CORRUPTED"] = "FILE_CORRUPTED";
|
|
23
|
-
/** File could not be found at the specified path */
|
|
24
|
-
OfficeErrorType["FILE_DOES_NOT_EXIST"] = "FILE_DOES_NOT_EXIST";
|
|
25
|
-
/** Specified location/directory is not reachable or is a directory */
|
|
26
|
-
OfficeErrorType["LOCATION_NOT_FOUND"] = "LOCATION_NOT_FOUND";
|
|
27
|
-
/** Arguments passed to the function are missing or invalid */
|
|
28
|
-
OfficeErrorType["IMPROPER_ARGUMENTS"] = "IMPROPER_ARGUMENTS";
|
|
29
|
-
/** Error occurred while reading or processing file buffers */
|
|
30
|
-
OfficeErrorType["IMPROPER_BUFFERS"] = "IMPROPER_BUFFERS";
|
|
31
|
-
/** Input type is not a supported type (string, Buffer, ArrayBuffer) */
|
|
32
|
-
OfficeErrorType["INVALID_INPUT"] = "INVALID_INPUT";
|
|
33
|
-
/** PDF worker source is missing (required in browser) */
|
|
34
|
-
OfficeErrorType["PDF_WORKER_MISSING"] = "PDF_WORKER_MISSING";
|
|
35
|
-
})(OfficeErrorType || (exports.OfficeErrorType = OfficeErrorType = {}));
|
|
36
14
|
/**
|
|
37
15
|
* Lookup table for error messages.
|
|
38
16
|
* Some entries are functions that take parameters to build dynamic messages.
|
|
39
17
|
*/
|
|
40
18
|
const ERROR_MESSAGES = {
|
|
41
|
-
[OfficeErrorType.EXTENSION_UNSUPPORTED]: (ext) => `Sorry, OfficeParser currently supports docx, pptx, xlsx, odt, odp, ods, pdf, rtf files only. Create a ticket in Issues on github to add support for ${ext} files. Stay tuned for further updates.`,
|
|
42
|
-
[OfficeErrorType.FILE_CORRUPTED]: (filepath) => `Your file ${filepath} seems to be corrupted. If you are sure it is fine, please create a ticket in Issues on github with the file to reproduce error.`,
|
|
43
|
-
[OfficeErrorType.FILE_DOES_NOT_EXIST]: (filepath) => `File ${filepath} could not be found! Check if the file exists or verify if the relative path to the file is correct from your terminal's location.`,
|
|
44
|
-
[OfficeErrorType.LOCATION_NOT_FOUND]: (location) => `Entered location ${location} is not reachable! Please make sure that the entered directory location exists. Check relative paths and reenter.`,
|
|
45
|
-
[OfficeErrorType.IMPROPER_ARGUMENTS]: `Improper arguments`,
|
|
46
|
-
[OfficeErrorType.IMPROPER_BUFFERS]: `
|
|
47
|
-
[OfficeErrorType.INVALID_INPUT]: `Invalid input type: Expected a Buffer or a valid file path`,
|
|
48
|
-
[OfficeErrorType.PDF_WORKER_MISSING]: `Missing PDF worker configuration. PDF parsing in browser environments requires a worker source. Please provide "pdfWorkerSrc" in your configuration
|
|
19
|
+
[types_js_1.OfficeErrorType.EXTENSION_UNSUPPORTED]: (ext) => `Sorry, OfficeParser currently supports docx, pptx, xlsx, odt, odp, ods, pdf, rtf, md, html, csv files only. Create a ticket in Issues on github to add support for ${ext} files. Stay tuned for further updates.`,
|
|
20
|
+
[types_js_1.OfficeErrorType.FILE_CORRUPTED]: (filepath) => `Your file ${filepath} seems to be corrupted. If you are sure it is fine, please create a ticket in Issues on github with the file to reproduce error.`,
|
|
21
|
+
[types_js_1.OfficeErrorType.FILE_DOES_NOT_EXIST]: (filepath) => `File ${filepath} could not be found! Check if the file exists or verify if the relative path to the file is correct from your terminal's location.`,
|
|
22
|
+
[types_js_1.OfficeErrorType.LOCATION_NOT_FOUND]: (location) => `Entered location ${location} is not reachable! Please make sure that the entered directory location exists. Check relative paths and reenter.`,
|
|
23
|
+
[types_js_1.OfficeErrorType.IMPROPER_ARGUMENTS]: `Improper arguments`,
|
|
24
|
+
[types_js_1.OfficeErrorType.IMPROPER_BUFFERS]: `Auto-detection of file type from buffer failed. This can happen if the format lacks magic bytes (like md, html, or csv) or if the detection library is incompatible with your Node.js version. Please provide the 'fileType' hint in your configuration (e.g., { fileType: 'docx' }) to proceed.`,
|
|
25
|
+
[types_js_1.OfficeErrorType.INVALID_INPUT]: `Invalid input type: Expected a Buffer or a valid file path`,
|
|
26
|
+
[types_js_1.OfficeErrorType.PDF_WORKER_MISSING]: `Missing PDF worker configuration. PDF parsing in browser environments requires a worker source. Please provide "pdfWorkerSrc" in your configuration.`,
|
|
27
|
+
[types_js_1.OfficeErrorType.FEATURE_NOT_SUPPORTED_IN_BROWSER]: (feature) => `'${feature}' is not supported in the browser. Browser users must pass file content as Buffer or ArrayBuffer directly.`,
|
|
28
|
+
[types_js_1.OfficeErrorType.INVALID_STYLE_MAPPING]: (mapping) => `Invalid style mapping string: ${mapping}`,
|
|
29
|
+
[types_js_1.OfficeErrorType.INVALID_SELECTOR]: (selector) => `Invalid selector: ${selector}`,
|
|
30
|
+
[types_js_1.OfficeErrorType.INVALID_OUTPUT_MAPPING]: (output) => `Invalid output mapping: ${output}`,
|
|
31
|
+
[types_js_1.OfficeErrorType.MISSING_EMBEDDING_FUNCTION]: `Semantic chunking requires an "embeddingFunction" to be provided in chunksConfig. This function must accept a string and return a Promise resolving to a number array (vector).`
|
|
49
32
|
};
|
|
33
|
+
/**
|
|
34
|
+
* Lookup table for warning messages.
|
|
35
|
+
*/
|
|
36
|
+
const WARNING_MESSAGES = {
|
|
37
|
+
[types_js_1.OfficeWarningType.PERFORMANCE_TIP]: (tip) => `⚡️ Performance Tip: ${tip}`,
|
|
38
|
+
[types_js_1.OfficeWarningType.OCR_FAILED]: (name) => `OCR failed for ${name}:`,
|
|
39
|
+
[types_js_1.OfficeWarningType.CHART_DATA_EXTRACTION_FAILED]: (path) => `Failed to extract chart data from ${path}:`,
|
|
40
|
+
[types_js_1.OfficeWarningType.PDF_WORKER_FALLBACK]: `Could not auto-resolve local worker path, falling back to CDN:`,
|
|
41
|
+
[types_js_1.OfficeWarningType.ATTACHMENT_EXTRACTION_FAILED]: `Error extracting embedded attachments:`,
|
|
42
|
+
[types_js_1.OfficeWarningType.PAGE_LOAD_FAILED]: (page) => `Error loading page ${page}:`,
|
|
43
|
+
[types_js_1.OfficeWarningType.DEPENDENCY_LOAD_FAILED]: (dep) => `Failed to load dependency ${dep}:`,
|
|
44
|
+
[types_js_1.OfficeWarningType.IMAGE_EXTRACTION_FAILED]: (context) => `Error extracting images ${context}:`,
|
|
45
|
+
[types_js_1.OfficeWarningType.ANNOTATION_EXTRACTION_FAILED]: (page) => `Error extracting annotations from page ${page}:`,
|
|
46
|
+
[types_js_1.OfficeWarningType.IMAGE_PROCESSING_FAILED]: `Failed to extract from ImageBitmap:`,
|
|
47
|
+
[types_js_1.OfficeWarningType.BROWSER_GENERATION_LIMITATION]: (msg) => msg,
|
|
48
|
+
[types_js_1.OfficeWarningType.SHEET_RANGE_NOT_FOUND]: (range) => `No sheets found matching the range: ${range}`,
|
|
49
|
+
[types_js_1.OfficeWarningType.BUFFER_TYPE_MISMATCH]: (info) => `File content type mismatch: Detected '${info.detected}' but expected/provided '${info.expected}'. Parsing will proceed with '${info.expected}' as requested.`,
|
|
50
|
+
[types_js_1.OfficeWarningType.FILE_TYPE_DETECTION_FAILED]: `Auto-detection of file type failed. This can happen on older Node.js versions with modern file-type versions. Please provide the 'fileType' hint in the configuration if parsing fails.`,
|
|
51
|
+
[types_js_1.OfficeWarningType.EMPTY_CHUNK_GENERATED]: (strategy) => `No chunks generated for document. Check if the document content is compatible with the '${strategy}' strategy.`,
|
|
52
|
+
[types_js_1.OfficeWarningType.WHITESPACE_NODE_SKIPPED]: (nodeType) => `Skipped whitespace-only node of type: ${nodeType}`
|
|
53
|
+
};
|
|
54
|
+
/**
|
|
55
|
+
* Creates a formatted warning message for a specific warning type.
|
|
56
|
+
*
|
|
57
|
+
* @param type - The type of warning
|
|
58
|
+
* @param info - Optional additional information
|
|
59
|
+
* @returns The formatted warning message string
|
|
60
|
+
*/
|
|
61
|
+
const getWarningMessage = (type, info) => {
|
|
62
|
+
const msg = WARNING_MESSAGES[type];
|
|
63
|
+
const message = typeof msg === 'function' ? msg(info) : msg;
|
|
64
|
+
return message;
|
|
65
|
+
};
|
|
66
|
+
exports.getWarningMessage = getWarningMessage;
|
|
50
67
|
/**
|
|
51
68
|
* Creates a formatted error message for a specific error type.
|
|
52
69
|
*
|
|
@@ -59,19 +76,41 @@ const createOfficeError = (type, info) => {
|
|
|
59
76
|
const message = typeof msg === 'function' ? msg(info) : msg;
|
|
60
77
|
return message;
|
|
61
78
|
};
|
|
79
|
+
/**
|
|
80
|
+
* Core reporting logic for all issues.
|
|
81
|
+
* Ensures consistent logging and callback execution.
|
|
82
|
+
*/
|
|
83
|
+
const reportIssue = (issue, config) => {
|
|
84
|
+
if (config?.onWarning) {
|
|
85
|
+
config.onWarning(issue);
|
|
86
|
+
}
|
|
87
|
+
else if (!config || config.outputErrorToConsole) {
|
|
88
|
+
const formatted = ERRORHEADER + issue.message;
|
|
89
|
+
if (issue.type === 'error') {
|
|
90
|
+
console.error(formatted, issue.details || '');
|
|
91
|
+
}
|
|
92
|
+
else {
|
|
93
|
+
console.warn(formatted, issue.details || '');
|
|
94
|
+
}
|
|
95
|
+
}
|
|
96
|
+
};
|
|
62
97
|
/**
|
|
63
98
|
* Creates, optionally logs to console, and returns a formatted OfficeParser error.
|
|
64
99
|
*
|
|
65
100
|
* @param type - The type of error
|
|
66
|
-
* @param config -
|
|
101
|
+
* @param config - Optional parser configuration (checks outputErrorToConsole)
|
|
67
102
|
* @param info - Optional additional information
|
|
68
103
|
* @returns The Error object to be thrown
|
|
69
104
|
*/
|
|
70
105
|
const getOfficeError = (type, config, info) => {
|
|
71
106
|
const message = createOfficeError(type, info);
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
107
|
+
const issue = {
|
|
108
|
+
type: 'error',
|
|
109
|
+
code: type,
|
|
110
|
+
message,
|
|
111
|
+
details: info
|
|
112
|
+
};
|
|
113
|
+
reportIssue(issue, config);
|
|
75
114
|
return new Error(ERRORHEADER + message);
|
|
76
115
|
};
|
|
77
116
|
exports.getOfficeError = getOfficeError;
|
|
@@ -86,35 +125,54 @@ exports.getOfficeError = getOfficeError;
|
|
|
86
125
|
*/
|
|
87
126
|
const getWrappedError = (error, config, filePath) => {
|
|
88
127
|
let message = error.message || error;
|
|
128
|
+
let code = types_js_1.OfficeErrorType.FILE_CORRUPTED; // Default for wrapped errors
|
|
89
129
|
// Detect file corruption from common library error messages
|
|
90
130
|
if (filePath && (message.includes('end of central directory record') ||
|
|
91
131
|
message.includes('invalid XML') ||
|
|
92
132
|
message.includes('Failed to open zip file') ||
|
|
93
133
|
message.includes('invalid distance too far back'))) {
|
|
94
|
-
message = createOfficeError(OfficeErrorType.FILE_CORRUPTED, filePath);
|
|
95
|
-
}
|
|
96
|
-
if (config.outputErrorToConsole) {
|
|
97
|
-
console.error(ERRORHEADER + message);
|
|
134
|
+
message = createOfficeError(types_js_1.OfficeErrorType.FILE_CORRUPTED, filePath);
|
|
98
135
|
}
|
|
136
|
+
const issue = {
|
|
137
|
+
type: 'error',
|
|
138
|
+
code: types_js_1.OfficeErrorType.FILE_CORRUPTED,
|
|
139
|
+
message,
|
|
140
|
+
details: filePath ? { filePath, originalError: error } : error
|
|
141
|
+
};
|
|
142
|
+
reportIssue(issue, config);
|
|
99
143
|
return new Error(ERRORHEADER + message);
|
|
100
144
|
};
|
|
101
145
|
exports.getWrappedError = getWrappedError;
|
|
102
146
|
/**
|
|
103
|
-
*
|
|
104
|
-
*
|
|
147
|
+
* Centralized logging utility for non-fatal warnings or issues.
|
|
148
|
+
* Routes messages to config.onWarning if provided, or console.warn/error
|
|
149
|
+
* if config.outputErrorToConsole is true.
|
|
105
150
|
*
|
|
106
|
-
* @param
|
|
107
|
-
* @param config -
|
|
108
|
-
* @param
|
|
151
|
+
* @param messageOrType - The warning message or warning type
|
|
152
|
+
* @param config - Optional parser configuration
|
|
153
|
+
* @param info - Optional additional information for dynamic messages or context
|
|
154
|
+
* @param error - Optional original error object
|
|
109
155
|
*/
|
|
110
|
-
const logWarning = (
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
156
|
+
const logWarning = (type, config, info, error) => {
|
|
157
|
+
let message;
|
|
158
|
+
let details = info;
|
|
159
|
+
const msg = WARNING_MESSAGES[type];
|
|
160
|
+
if (typeof msg === 'function') {
|
|
161
|
+
message = msg(info);
|
|
162
|
+
details = error || info;
|
|
163
|
+
}
|
|
164
|
+
else {
|
|
165
|
+
message = msg;
|
|
166
|
+
if (info instanceof Error && !error) {
|
|
167
|
+
details = info;
|
|
117
168
|
}
|
|
118
169
|
}
|
|
170
|
+
const issue = {
|
|
171
|
+
type: 'warning',
|
|
172
|
+
code: type,
|
|
173
|
+
message,
|
|
174
|
+
details
|
|
175
|
+
};
|
|
176
|
+
reportIssue(issue, config);
|
|
119
177
|
};
|
|
120
178
|
exports.logWarning = logWarning;
|
|
@@ -12,22 +12,30 @@ Object.defineProperty(exports, "__esModule", { value: true });
|
|
|
12
12
|
exports.loadFileType = loadFileType;
|
|
13
13
|
exports.loadPdfJs = loadPdfJs;
|
|
14
14
|
const envUtils_js_1 = require("./envUtils.js");
|
|
15
|
-
/**
|
|
16
|
-
* Dynamically loads an ESM module in a Node.js CJS context.
|
|
17
|
-
*
|
|
18
|
-
* @param specifier - The module specifier to load
|
|
19
|
-
* @returns The loaded module
|
|
20
|
-
*/
|
|
21
15
|
async function loadNodeEsmModule(specifier) {
|
|
22
|
-
//
|
|
23
|
-
// This ensures that
|
|
24
|
-
|
|
16
|
+
// In Node.js, we resolve the specifier to an absolute file URL.
|
|
17
|
+
// This ensures that the dynamic import() call (executed via new Function)
|
|
18
|
+
// always finds the correct module regardless of the caller's context.
|
|
19
|
+
// This is especially important in Node 18 for sub-paths of packages.
|
|
20
|
+
try {
|
|
21
|
+
const { pathToFileURL } = await import('url');
|
|
22
|
+
// @ts-ignore - require.resolve is available in Node.js
|
|
23
|
+
const absolutePath = require.resolve(specifier);
|
|
24
|
+
const fileUrl = pathToFileURL(absolutePath).href;
|
|
25
|
+
return new Function('s', 'return import(s)')(fileUrl);
|
|
26
|
+
}
|
|
27
|
+
catch (e) {
|
|
28
|
+
// Fallback for cases where require.resolve might fail (e.g. non-file specifiers)
|
|
29
|
+
return new Function('s', 'return import(s)')(specifier);
|
|
30
|
+
}
|
|
25
31
|
}
|
|
26
32
|
/**
|
|
27
33
|
* Specialized loader for file-type
|
|
28
34
|
*/
|
|
29
35
|
async function loadFileType() {
|
|
30
36
|
if (!envUtils_js_1.isBrowser) {
|
|
37
|
+
// Ensure environment polyfills for Node.js 18 support
|
|
38
|
+
(0, envUtils_js_1.ensureEnvPolyfills)();
|
|
31
39
|
// Node.js path: Use dynamic import wrapper for CJS compatibility
|
|
32
40
|
return loadNodeEsmModule('file-type');
|
|
33
41
|
}
|
|
@@ -39,8 +47,8 @@ async function loadFileType() {
|
|
|
39
47
|
*/
|
|
40
48
|
async function loadPdfJs() {
|
|
41
49
|
if (!envUtils_js_1.isBrowser) {
|
|
42
|
-
// Ensure
|
|
43
|
-
(0, envUtils_js_1.
|
|
50
|
+
// Ensure environment polyfills for Node.js 18 support
|
|
51
|
+
(0, envUtils_js_1.ensureEnvPolyfills)();
|
|
44
52
|
// Node.js environment: require legacy build for stability with ESM-only main
|
|
45
53
|
try {
|
|
46
54
|
return await loadNodeEsmModule('pdfjs-dist/legacy/build/pdf.mjs');
|
package/dist/utils/ocrUtils.js
CHANGED
|
@@ -12,6 +12,7 @@
|
|
|
12
12
|
*/
|
|
13
13
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
14
14
|
exports.terminateOcr = exports.performOcr = void 0;
|
|
15
|
+
const envUtils_js_1 = require("./envUtils.js");
|
|
15
16
|
/**
|
|
16
17
|
* Manages a pool of Tesseract workers with "Smart Affinity".
|
|
17
18
|
*
|
|
@@ -202,7 +203,7 @@ const performOcr = async (image, config) => {
|
|
|
202
203
|
let inputImage = image;
|
|
203
204
|
// In browser environment, convert Buffer to Blob for better compatibility
|
|
204
205
|
// @ts-ignore
|
|
205
|
-
if (
|
|
206
|
+
if (envUtils_js_1.isBrowser && typeof Blob !== 'undefined' && Buffer.isBuffer(image)) {
|
|
206
207
|
inputImage = new Blob([image], { type: 'image/bmp' });
|
|
207
208
|
}
|
|
208
209
|
return await OcrSchedulerManager.getInstance().recognize(inputImage, config);
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Parses a range string (e.g., "1", "1-3", "1,2", "1,3-5, 7") into an array of numbers.
|
|
3
|
+
*
|
|
4
|
+
* @param rangeStr - The range string to parse
|
|
5
|
+
* @returns An array of unique, sorted numbers (1-based indices)
|
|
6
|
+
*/
|
|
7
|
+
export declare function parseRangeString(rangeStr: string): number[];
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.parseRangeString = parseRangeString;
|
|
4
|
+
/**
|
|
5
|
+
* Parses a range string (e.g., "1", "1-3", "1,2", "1,3-5, 7") into an array of numbers.
|
|
6
|
+
*
|
|
7
|
+
* @param rangeStr - The range string to parse
|
|
8
|
+
* @returns An array of unique, sorted numbers (1-based indices)
|
|
9
|
+
*/
|
|
10
|
+
function parseRangeString(rangeStr) {
|
|
11
|
+
const result = new Set();
|
|
12
|
+
const segments = rangeStr.split(',');
|
|
13
|
+
for (const segment of segments) {
|
|
14
|
+
const trimmed = segment.trim();
|
|
15
|
+
if (trimmed.includes('-')) {
|
|
16
|
+
const [startStr, endStr] = trimmed.split('-');
|
|
17
|
+
const start = parseInt(startStr, 10);
|
|
18
|
+
const end = parseInt(endStr, 10);
|
|
19
|
+
if (!isNaN(start) && !isNaN(end)) {
|
|
20
|
+
const actualStart = Math.min(start, end);
|
|
21
|
+
const actualEnd = Math.max(start, end);
|
|
22
|
+
for (let i = actualStart; i <= actualEnd; i++) {
|
|
23
|
+
result.add(i);
|
|
24
|
+
}
|
|
25
|
+
}
|
|
26
|
+
}
|
|
27
|
+
else {
|
|
28
|
+
const val = parseInt(trimmed, 10);
|
|
29
|
+
if (!isNaN(val)) {
|
|
30
|
+
result.add(val);
|
|
31
|
+
}
|
|
32
|
+
}
|
|
33
|
+
}
|
|
34
|
+
return Array.from(result).sort((a, b) => a - b);
|
|
35
|
+
}
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
import { OfficeContentNode, StructuredStyleMapping } from '../types.js';
|
|
2
|
+
export interface StyleMapping {
|
|
3
|
+
selector: {
|
|
4
|
+
nodeType?: string;
|
|
5
|
+
attributes: Record<string, {
|
|
6
|
+
value: string | number | boolean;
|
|
7
|
+
operator: '=' | '~=';
|
|
8
|
+
compiled?: RegExp;
|
|
9
|
+
}>;
|
|
10
|
+
};
|
|
11
|
+
output: {
|
|
12
|
+
tag: string;
|
|
13
|
+
classes: string[];
|
|
14
|
+
attributes: Record<string, string>;
|
|
15
|
+
fresh: boolean;
|
|
16
|
+
};
|
|
17
|
+
}
|
|
18
|
+
/**
|
|
19
|
+
* Parser and matcher for the style mapping DSL.
|
|
20
|
+
* Supports a structured JSON format and a legacy string DSL.
|
|
21
|
+
*/
|
|
22
|
+
export declare class StyleMapper {
|
|
23
|
+
private mappings;
|
|
24
|
+
constructor(mappings?: string[] | StructuredStyleMapping[] | Record<string, any>, ignoreDefaults?: boolean);
|
|
25
|
+
/**
|
|
26
|
+
* Finds the best matching mapping for a node.
|
|
27
|
+
*/
|
|
28
|
+
getMapping(node: OfficeContentNode): StyleMapping['output'] | undefined;
|
|
29
|
+
private matches;
|
|
30
|
+
private getNodeAttribute;
|
|
31
|
+
private convertStructuredMapping;
|
|
32
|
+
/**
|
|
33
|
+
* Parses a mapping string like "p[style-name='Heading 1'] => h1.title:fresh"
|
|
34
|
+
*/
|
|
35
|
+
private parseMappingString;
|
|
36
|
+
}
|
|
@@ -0,0 +1,224 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.StyleMapper = void 0;
|
|
4
|
+
const types_js_1 = require("../types.js");
|
|
5
|
+
const errorUtils_js_1 = require("./errorUtils.js");
|
|
6
|
+
const DEFAULT_MAPPINGS = [
|
|
7
|
+
{ selector: { nodeType: 'paragraph', attributes: { 'style-name': 'Heading 1' } }, output: { tag: 'h1' } },
|
|
8
|
+
{ selector: { nodeType: 'paragraph', attributes: { 'style-name': 'Heading 2' } }, output: { tag: 'h2' } },
|
|
9
|
+
{ selector: { nodeType: 'paragraph', attributes: { 'style-name': 'Heading 3' } }, output: { tag: 'h3' } },
|
|
10
|
+
{ selector: { nodeType: 'paragraph', attributes: { 'style-name': 'Heading 4' } }, output: { tag: 'h4' } },
|
|
11
|
+
{ selector: { nodeType: 'paragraph', attributes: { 'style-name': 'Heading 5' } }, output: { tag: 'h5' } },
|
|
12
|
+
{ selector: { nodeType: 'paragraph', attributes: { 'style-name': 'Heading 6' } }, output: { tag: 'h6' } },
|
|
13
|
+
{ selector: { nodeType: 'paragraph', attributes: { 'style-name': 'Title' } }, output: { tag: 'h1', classes: ['title'] } },
|
|
14
|
+
{ selector: { nodeType: 'paragraph', attributes: { 'style-name': 'Subtitle' } }, output: { tag: 'p', classes: ['subtitle'] } },
|
|
15
|
+
{ selector: { nodeType: 'paragraph', attributes: { 'style-name': 'Quote' } }, output: { tag: 'blockquote' } },
|
|
16
|
+
{ selector: { nodeType: 'paragraph', attributes: { 'style-name': 'Intense Quote' } }, output: { tag: 'blockquote', classes: ['intense'] } },
|
|
17
|
+
];
|
|
18
|
+
/**
|
|
19
|
+
* Parser and matcher for the style mapping DSL.
|
|
20
|
+
* Supports a structured JSON format and a legacy string DSL.
|
|
21
|
+
*/
|
|
22
|
+
class StyleMapper {
|
|
23
|
+
mappings = [];
|
|
24
|
+
constructor(mappings, ignoreDefaults = false) {
|
|
25
|
+
// 1. Add user mappings (they take precedence)
|
|
26
|
+
if (mappings) {
|
|
27
|
+
if (Array.isArray(mappings)) {
|
|
28
|
+
for (const m of mappings) {
|
|
29
|
+
if (typeof m === 'string') {
|
|
30
|
+
this.mappings.push(this.parseMappingString(m));
|
|
31
|
+
}
|
|
32
|
+
else {
|
|
33
|
+
this.mappings.push(this.convertStructuredMapping(m));
|
|
34
|
+
}
|
|
35
|
+
}
|
|
36
|
+
}
|
|
37
|
+
else {
|
|
38
|
+
// Support legacy object format: { 'Heading 1': { tag: 'h1', class: 'title' } }
|
|
39
|
+
for (const [styleName, target] of Object.entries(mappings)) {
|
|
40
|
+
this.mappings.push({
|
|
41
|
+
selector: {
|
|
42
|
+
attributes: { style: { value: styleName, operator: '=' } }
|
|
43
|
+
},
|
|
44
|
+
output: {
|
|
45
|
+
tag: target.tag || 'div',
|
|
46
|
+
classes: target.class ? target.class.split(' ') : [],
|
|
47
|
+
attributes: {},
|
|
48
|
+
fresh: false
|
|
49
|
+
}
|
|
50
|
+
});
|
|
51
|
+
}
|
|
52
|
+
}
|
|
53
|
+
}
|
|
54
|
+
// 2. Add default mappings if not ignored
|
|
55
|
+
if (!ignoreDefaults) {
|
|
56
|
+
this.mappings.push(...DEFAULT_MAPPINGS.map(m => this.convertStructuredMapping(m)));
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
/**
|
|
60
|
+
* Finds the best matching mapping for a node.
|
|
61
|
+
*/
|
|
62
|
+
getMapping(node) {
|
|
63
|
+
for (const mapping of this.mappings) {
|
|
64
|
+
if (this.matches(node, mapping.selector)) {
|
|
65
|
+
return mapping.output;
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
return undefined;
|
|
69
|
+
}
|
|
70
|
+
matches(node, selector) {
|
|
71
|
+
// Match node type if specified
|
|
72
|
+
if (selector.nodeType && node.type !== selector.nodeType) {
|
|
73
|
+
return false;
|
|
74
|
+
}
|
|
75
|
+
// Match attributes (style, level, etc.)
|
|
76
|
+
for (const [attr, { value, operator, compiled }] of Object.entries(selector.attributes)) {
|
|
77
|
+
const actualValue = this.getNodeAttribute(node, attr);
|
|
78
|
+
if (actualValue === undefined)
|
|
79
|
+
return false;
|
|
80
|
+
if (operator === '=') {
|
|
81
|
+
if (String(actualValue) !== String(value))
|
|
82
|
+
return false;
|
|
83
|
+
}
|
|
84
|
+
else if (operator === '~=') {
|
|
85
|
+
const regex = compiled || new RegExp(String(value));
|
|
86
|
+
if (!regex.test(String(actualValue)))
|
|
87
|
+
return false;
|
|
88
|
+
}
|
|
89
|
+
}
|
|
90
|
+
return true;
|
|
91
|
+
}
|
|
92
|
+
getNodeAttribute(node, attr) {
|
|
93
|
+
// Special case for style (alias style-name for mammoth.js compatibility)
|
|
94
|
+
if (attr === 'style' || attr === 'style-name') {
|
|
95
|
+
return node.metadata?.style || node.formatting?.font;
|
|
96
|
+
}
|
|
97
|
+
// Metadata attributes
|
|
98
|
+
if (node.metadata && attr in node.metadata) {
|
|
99
|
+
return node.metadata[attr];
|
|
100
|
+
}
|
|
101
|
+
// Formatting attributes
|
|
102
|
+
if (node.formatting && attr in node.formatting) {
|
|
103
|
+
return node.formatting[attr];
|
|
104
|
+
}
|
|
105
|
+
return undefined;
|
|
106
|
+
}
|
|
107
|
+
convertStructuredMapping(m) {
|
|
108
|
+
const attributes = {};
|
|
109
|
+
if (m.selector.attributes) {
|
|
110
|
+
for (const [key, val] of Object.entries(m.selector.attributes)) {
|
|
111
|
+
if (typeof val === 'object' && val !== null && 'value' in val) {
|
|
112
|
+
const operator = val.operator || '=';
|
|
113
|
+
attributes[key] = {
|
|
114
|
+
value: val.value,
|
|
115
|
+
operator,
|
|
116
|
+
compiled: operator === '~=' ? new RegExp(String(val.value)) : undefined
|
|
117
|
+
};
|
|
118
|
+
}
|
|
119
|
+
else {
|
|
120
|
+
attributes[key] = {
|
|
121
|
+
value: val,
|
|
122
|
+
operator: '='
|
|
123
|
+
};
|
|
124
|
+
}
|
|
125
|
+
}
|
|
126
|
+
}
|
|
127
|
+
return {
|
|
128
|
+
selector: {
|
|
129
|
+
nodeType: m.selector.nodeType,
|
|
130
|
+
attributes
|
|
131
|
+
},
|
|
132
|
+
output: {
|
|
133
|
+
tag: m.output.tag,
|
|
134
|
+
classes: m.output.classes || [],
|
|
135
|
+
attributes: m.output.attributes || {},
|
|
136
|
+
fresh: m.output.fresh || false
|
|
137
|
+
}
|
|
138
|
+
};
|
|
139
|
+
}
|
|
140
|
+
/**
|
|
141
|
+
* Parses a mapping string like "p[style-name='Heading 1'] => h1.title:fresh"
|
|
142
|
+
*/
|
|
143
|
+
parseMappingString(mapping) {
|
|
144
|
+
const lastIndex = mapping.lastIndexOf('=>');
|
|
145
|
+
if (lastIndex === -1) {
|
|
146
|
+
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.INVALID_STYLE_MAPPING, undefined, mapping);
|
|
147
|
+
}
|
|
148
|
+
const selectorStr = mapping.substring(0, lastIndex).trim();
|
|
149
|
+
const outputStr = mapping.substring(lastIndex + 2).trim();
|
|
150
|
+
// Parse Selector
|
|
151
|
+
const selectorMatch = selectorStr.match(/^([a-z]+)?(?:\[(.+?)\])?$/);
|
|
152
|
+
if (!selectorMatch) {
|
|
153
|
+
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.INVALID_SELECTOR, undefined, selectorStr);
|
|
154
|
+
}
|
|
155
|
+
const typeMap = {
|
|
156
|
+
'p': 'paragraph',
|
|
157
|
+
'h': 'heading',
|
|
158
|
+
't': 'table',
|
|
159
|
+
'tr': 'row',
|
|
160
|
+
'td': 'cell',
|
|
161
|
+
'li': 'list',
|
|
162
|
+
'img': 'image'
|
|
163
|
+
};
|
|
164
|
+
const nodeType = selectorMatch[1] ? (typeMap[selectorMatch[1]] || selectorMatch[1]) : undefined;
|
|
165
|
+
const attrStr = selectorMatch[2];
|
|
166
|
+
const attributes = {};
|
|
167
|
+
if (attrStr) {
|
|
168
|
+
// Improved attribute parsing to handle commas inside quotes
|
|
169
|
+
const attrParts = [];
|
|
170
|
+
let currentPart = '';
|
|
171
|
+
let inQuotes = false;
|
|
172
|
+
for (let i = 0; i < attrStr.length; i++) {
|
|
173
|
+
const char = attrStr[i];
|
|
174
|
+
if (char === "'" || char === '"')
|
|
175
|
+
inQuotes = !inQuotes;
|
|
176
|
+
if (char === ',' && !inQuotes) {
|
|
177
|
+
attrParts.push(currentPart.trim());
|
|
178
|
+
currentPart = '';
|
|
179
|
+
}
|
|
180
|
+
else {
|
|
181
|
+
currentPart += char;
|
|
182
|
+
}
|
|
183
|
+
}
|
|
184
|
+
if (currentPart)
|
|
185
|
+
attrParts.push(currentPart.trim());
|
|
186
|
+
for (const part of attrParts) {
|
|
187
|
+
const m = part.match(/^([\w-]+)\s*(=|~=)\s*(?:(["'])(.*?)\3|(.+))$/);
|
|
188
|
+
if (m) {
|
|
189
|
+
const operator = m[2];
|
|
190
|
+
const value = m[4] !== undefined ? m[4] : m[5];
|
|
191
|
+
attributes[m[1]] = {
|
|
192
|
+
operator,
|
|
193
|
+
value,
|
|
194
|
+
compiled: operator === '~=' ? new RegExp(value) : undefined
|
|
195
|
+
};
|
|
196
|
+
}
|
|
197
|
+
}
|
|
198
|
+
}
|
|
199
|
+
// Parse Output
|
|
200
|
+
const outputParts = outputStr.split(':');
|
|
201
|
+
const fresh = outputParts.includes('fresh');
|
|
202
|
+
const mainOutput = outputParts[0];
|
|
203
|
+
const outputMatch = mainOutput.match(/^([a-z0-9]+)?((?:\.[\w-]+)*)(?:\[(.+?)\])?$/);
|
|
204
|
+
if (!outputMatch) {
|
|
205
|
+
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.INVALID_OUTPUT_MAPPING, undefined, mainOutput);
|
|
206
|
+
}
|
|
207
|
+
const tag = outputMatch[1] || 'div';
|
|
208
|
+
const classes = outputMatch[2] ? outputMatch[2].split('.').filter(Boolean) : [];
|
|
209
|
+
const outAttrs = {};
|
|
210
|
+
if (outputMatch[3]) {
|
|
211
|
+
const outAttrParts = outputMatch[3].split(',').map(a => a.trim());
|
|
212
|
+
for (const part of outAttrParts) {
|
|
213
|
+
const m = part.match(/^([\w-]+)\s*=\s*(?:(["'])(.*?)\2|(.+))$/);
|
|
214
|
+
if (m)
|
|
215
|
+
outAttrs[m[1]] = m[3] !== undefined ? m[3] : m[4];
|
|
216
|
+
}
|
|
217
|
+
}
|
|
218
|
+
return {
|
|
219
|
+
selector: { nodeType, attributes },
|
|
220
|
+
output: { tag, classes, attributes: outAttrs, fresh }
|
|
221
|
+
};
|
|
222
|
+
}
|
|
223
|
+
}
|
|
224
|
+
exports.StyleMapper = StyleMapper;
|
package/dist/utils/xmlUtils.d.ts
CHANGED
|
@@ -42,14 +42,6 @@ export declare const parseXmlString: (xml: string, options?: {
|
|
|
42
42
|
* ```
|
|
43
43
|
*/
|
|
44
44
|
export declare const getElementsByTagName: (element: Element | Document, tagName: string) => Element[];
|
|
45
|
-
/**
|
|
46
|
-
* Serializes a DOM Node (Document, Element, etc.) back into an XML string.
|
|
47
|
-
* This is cross-platform and works in both Node.js and Browser environments.
|
|
48
|
-
*
|
|
49
|
-
* @param node - The DOM node to serialize
|
|
50
|
-
* @param options - Serialization options
|
|
51
|
-
* @returns The XML string representation
|
|
52
|
-
*/
|
|
53
45
|
export declare const serializeXml: (node: Node, options?: {
|
|
54
46
|
preserveWhitespace?: boolean;
|
|
55
47
|
}) => string;
|
package/dist/utils/xmlUtils.js
CHANGED
|
@@ -72,13 +72,14 @@ exports.getElementsByTagName = getElementsByTagName;
|
|
|
72
72
|
* @param options - Serialization options
|
|
73
73
|
* @returns The XML string representation
|
|
74
74
|
*/
|
|
75
|
+
const serializer = new xmldom_1.XMLSerializer();
|
|
75
76
|
const serializeXml = (node, options = {}) => {
|
|
76
77
|
// Note: xmldom's XMLSerializer doesn't natively support a 'pretty' or 'preserve'
|
|
77
78
|
// flag in a way that matches all user expectations, but it defaults to
|
|
78
79
|
// preserving structure. Formatting (indentation) is usually handled by the
|
|
79
80
|
// parser's initial whitespace handling.
|
|
80
81
|
// @ts-ignore - xmldom's Node is compatible with the global Node interface
|
|
81
|
-
return
|
|
82
|
+
return serializer.serializeToString(node);
|
|
82
83
|
};
|
|
83
84
|
exports.serializeXml = serializeXml;
|
|
84
85
|
/**
|