officeparser 5.2.1 → 6.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +411 -163
- package/dist/OfficeParser.d.ts +90 -0
- package/dist/OfficeParser.js +217 -0
- package/dist/index.d.ts +51 -0
- package/dist/index.js +108 -0
- package/dist/officeparser.browser.js +165 -0
- package/dist/officeparser.browser.js.map +7 -0
- package/dist/parsers/ExcelParser.d.ts +33 -0
- package/dist/parsers/ExcelParser.js +643 -0
- package/dist/parsers/OpenOfficeParser.d.ts +32 -0
- package/dist/parsers/OpenOfficeParser.js +1399 -0
- package/dist/parsers/PdfParser.d.ts +68 -0
- package/dist/parsers/PdfParser.js +847 -0
- package/dist/parsers/PowerPointParser.d.ts +33 -0
- package/dist/parsers/PowerPointParser.js +778 -0
- package/dist/parsers/RtfParser.d.ts +164 -0
- package/dist/parsers/RtfParser.js +1641 -0
- package/dist/parsers/WordParser.d.ts +79 -0
- package/dist/parsers/WordParser.js +787 -0
- package/dist/types.d.ts +615 -0
- package/dist/types.js +2 -0
- package/dist/utils/chartUtils.d.ts +7 -0
- package/dist/utils/chartUtils.js +255 -0
- package/dist/utils/errorUtils.d.ts +58 -0
- package/dist/utils/errorUtils.js +120 -0
- package/dist/utils/imageUtils.d.ts +67 -0
- package/dist/utils/imageUtils.js +133 -0
- package/dist/utils/ocrUtils.d.ts +39 -0
- package/dist/utils/ocrUtils.js +61 -0
- package/dist/utils/xmlUtils.d.ts +83 -0
- package/dist/utils/xmlUtils.js +158 -0
- package/dist/utils/zipUtils.d.ts +74 -0
- package/dist/utils/zipUtils.js +112 -0
- package/package.json +44 -17
- package/officeParser.js +0 -776
- package/typings/officeParser.d.ts +0 -33
|
@@ -0,0 +1,255 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.extractChartData = void 0;
|
|
4
|
+
const xmlUtils_1 = require("./xmlUtils");
|
|
5
|
+
/**
|
|
6
|
+
* Extracts a single text element located at:
|
|
7
|
+
* c:title -> c:tx -> c:rich -> a:p -> a:r -> a:t
|
|
8
|
+
* OR c:tx -> c:strRef -> c:strCache -> c:pt -> c:v
|
|
9
|
+
* @param el Chart element
|
|
10
|
+
* @param tagName The tag to search for (e.g., "c:title", "c:tx")
|
|
11
|
+
*/
|
|
12
|
+
const extractOpenXmlRichText = (el, tagName) => {
|
|
13
|
+
const target = (el.localName === tagName || el.tagName === tagName) ? el : el.getElementsByTagName(tagName)[0];
|
|
14
|
+
if (!target)
|
|
15
|
+
return undefined;
|
|
16
|
+
// 1. Try c:rich or a:p (standard rich text)
|
|
17
|
+
const richNodes = target.getElementsByTagName("c:rich");
|
|
18
|
+
const pNodes = target.getElementsByTagName("a:p");
|
|
19
|
+
const textContainers = richNodes.length > 0 ? Array.from(richNodes) : Array.from(pNodes);
|
|
20
|
+
if (textContainers.length > 0) {
|
|
21
|
+
let acc = "";
|
|
22
|
+
for (const container of textContainers) {
|
|
23
|
+
const tNodes = container.getElementsByTagName("a:t");
|
|
24
|
+
for (let i = 0; i < tNodes.length; i++) {
|
|
25
|
+
acc += (tNodes[i].textContent || "") + " ";
|
|
26
|
+
}
|
|
27
|
+
}
|
|
28
|
+
if (acc.trim())
|
|
29
|
+
return acc.trim();
|
|
30
|
+
}
|
|
31
|
+
// 2. Try c:v (cached values/strings)
|
|
32
|
+
const vNode = target.getElementsByTagName("c:v")[0];
|
|
33
|
+
if (vNode && vNode.textContent) {
|
|
34
|
+
return vNode.textContent.trim() || undefined;
|
|
35
|
+
}
|
|
36
|
+
return undefined;
|
|
37
|
+
};
|
|
38
|
+
/**
|
|
39
|
+
* Extracts fully structured chart data from OpenXML (PPTX, XLSX) chart XML.
|
|
40
|
+
* @param xmlBuffer Chart XML buffer
|
|
41
|
+
*/
|
|
42
|
+
const extractOpenXmlChartData = (xmlBuffer) => {
|
|
43
|
+
const xml = xmlBuffer.toString("utf8");
|
|
44
|
+
const dom = (0, xmlUtils_1.parseXmlString)(xml);
|
|
45
|
+
const root = dom.documentElement;
|
|
46
|
+
const title = extractOpenXmlRichText(root, "c:title");
|
|
47
|
+
// Axis titles
|
|
48
|
+
let xAxisTitle = undefined;
|
|
49
|
+
let yAxisTitle = undefined;
|
|
50
|
+
const catAxes = root.getElementsByTagName("c:catAx");
|
|
51
|
+
if (catAxes.length > 0)
|
|
52
|
+
xAxisTitle = extractOpenXmlRichText(catAxes[0], "c:title");
|
|
53
|
+
const valAxes = root.getElementsByTagName("c:valAx");
|
|
54
|
+
if (valAxes.length > 0)
|
|
55
|
+
yAxisTitle = extractOpenXmlRichText(valAxes[0], "c:title");
|
|
56
|
+
// Extract Series (dataSets)
|
|
57
|
+
const seriesNodes = root.getElementsByTagName("c:ser");
|
|
58
|
+
const dataSets = [];
|
|
59
|
+
const sharedLabels = [];
|
|
60
|
+
for (let i = 0; i < seriesNodes.length; i++) {
|
|
61
|
+
const ser = seriesNodes[i];
|
|
62
|
+
// DataSet Name (c:tx)
|
|
63
|
+
const name = extractOpenXmlRichText(ser, "c:tx");
|
|
64
|
+
// Values (c:val)
|
|
65
|
+
const values = [];
|
|
66
|
+
const valNode = ser.getElementsByTagName("c:val")[0] || ser.getElementsByTagName("c:yVal")[0];
|
|
67
|
+
if (valNode) {
|
|
68
|
+
const vNodes = valNode.getElementsByTagName("c:v");
|
|
69
|
+
for (let j = 0; j < vNodes.length; j++) {
|
|
70
|
+
const v = vNodes[j].textContent?.trim();
|
|
71
|
+
if (v)
|
|
72
|
+
values.push(v);
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
// Point Labels (data labels)
|
|
76
|
+
const pointLabels = [];
|
|
77
|
+
const dLbls = ser.getElementsByTagName("c:dLbl");
|
|
78
|
+
for (let j = 0; j < dLbls.length; j++) {
|
|
79
|
+
const lbl = extractOpenXmlRichText(dLbls[j], "c:tx");
|
|
80
|
+
if (lbl)
|
|
81
|
+
pointLabels.push(lbl);
|
|
82
|
+
}
|
|
83
|
+
// Categories (labels) - c:cat or c:xVal
|
|
84
|
+
const catNode = ser.getElementsByTagName("c:cat")[0] || ser.getElementsByTagName("c:xVal")[0];
|
|
85
|
+
if (catNode) {
|
|
86
|
+
const vNodes = catNode.getElementsByTagName("c:v");
|
|
87
|
+
const localLabels = [];
|
|
88
|
+
for (let j = 0; j < vNodes.length; j++) {
|
|
89
|
+
const v = vNodes[j].textContent?.trim();
|
|
90
|
+
if (v)
|
|
91
|
+
localLabels.push(v);
|
|
92
|
+
}
|
|
93
|
+
if (localLabels.length > 0 && sharedLabels.length === 0) {
|
|
94
|
+
sharedLabels.push(...localLabels);
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
dataSets.push({ name, values, pointLabels });
|
|
98
|
+
}
|
|
99
|
+
// Structured rawTexts: for each dataset: Name -> Labels -> Values
|
|
100
|
+
const rawTexts = [];
|
|
101
|
+
for (const ds of dataSets) {
|
|
102
|
+
if (ds.name)
|
|
103
|
+
rawTexts.push(ds.name);
|
|
104
|
+
rawTexts.push(...sharedLabels);
|
|
105
|
+
rawTexts.push(...ds.values);
|
|
106
|
+
}
|
|
107
|
+
return {
|
|
108
|
+
title,
|
|
109
|
+
xAxisTitle,
|
|
110
|
+
yAxisTitle,
|
|
111
|
+
dataSets,
|
|
112
|
+
labels: sharedLabels,
|
|
113
|
+
rawTexts
|
|
114
|
+
};
|
|
115
|
+
};
|
|
116
|
+
/**
|
|
117
|
+
* Extracts structured chart data from ODF (ODP, ODS) chart content.xml.
|
|
118
|
+
* @param xmlBuffer Chart XML buffer
|
|
119
|
+
*/
|
|
120
|
+
const extractOdfChartData = (xmlBuffer) => {
|
|
121
|
+
const xml = xmlBuffer.toString("utf8");
|
|
122
|
+
const dom = (0, xmlUtils_1.parseXmlString)(xml);
|
|
123
|
+
const chart = (0, xmlUtils_1.getElementsByTagName)(dom, "chart:chart")[0] || dom.documentElement;
|
|
124
|
+
const titleNode = (0, xmlUtils_1.getElementsByTagName)(chart, "chart:title")[0];
|
|
125
|
+
const title = titleNode ? (0, xmlUtils_1.getElementsByTagName)(titleNode, "text:p")[0]?.textContent || undefined : undefined;
|
|
126
|
+
const table = (0, xmlUtils_1.getElementsByTagName)(chart, "table:table")[0];
|
|
127
|
+
const dataSets = [];
|
|
128
|
+
const labels = [];
|
|
129
|
+
const rawTexts = [];
|
|
130
|
+
if (table) {
|
|
131
|
+
// Chart with embedded data table (common in ODP presentations)
|
|
132
|
+
let rows = [];
|
|
133
|
+
const headerRowsNode = (0, xmlUtils_1.getDirectChildren)(table, "table:table-header-rows")[0];
|
|
134
|
+
if (headerRowsNode) {
|
|
135
|
+
rows.push(...(0, xmlUtils_1.getDirectChildren)(headerRowsNode, "table:table-row"));
|
|
136
|
+
}
|
|
137
|
+
rows.push(...(0, xmlUtils_1.getDirectChildren)(table, "table:table-row"));
|
|
138
|
+
if (rows.length > 0) {
|
|
139
|
+
// Header row for series names
|
|
140
|
+
const headerCells = (0, xmlUtils_1.getDirectChildren)(rows[0], "table:table-cell");
|
|
141
|
+
for (let j = 1; j < headerCells.length; j++) {
|
|
142
|
+
const colsRepeated = parseInt(headerCells[j].getAttribute("table:number-columns-repeated") || "1");
|
|
143
|
+
const name = (0, xmlUtils_1.getDirectChildren)(headerCells[j], "text:p")[0]?.textContent || undefined;
|
|
144
|
+
for (let k = 0; k < colsRepeated; k++) {
|
|
145
|
+
dataSets.push({ name, values: [], pointLabels: [] });
|
|
146
|
+
}
|
|
147
|
+
}
|
|
148
|
+
// Data rows
|
|
149
|
+
for (let i = 1; i < rows.length; i++) {
|
|
150
|
+
const dataCells = (0, xmlUtils_1.getDirectChildren)(rows[i], "table:table-cell");
|
|
151
|
+
if (dataCells.length > 0) {
|
|
152
|
+
const label = (0, xmlUtils_1.getDirectChildren)(dataCells[0], "text:p")[0]?.textContent || undefined;
|
|
153
|
+
if (label)
|
|
154
|
+
labels.push(label);
|
|
155
|
+
let dsIdx = 0;
|
|
156
|
+
for (let j = 1; j < dataCells.length; j++) {
|
|
157
|
+
const colsRepeated = parseInt(dataCells[j].getAttribute("table:number-columns-repeated") || "1");
|
|
158
|
+
const val = dataCells[j].getAttribute("office:value") || (0, xmlUtils_1.getDirectChildren)(dataCells[j], "text:p")[0]?.textContent || "";
|
|
159
|
+
for (let k = 0; k < colsRepeated; k++) {
|
|
160
|
+
if (dataSets[dsIdx]) {
|
|
161
|
+
dataSets[dsIdx].values.push(val);
|
|
162
|
+
}
|
|
163
|
+
dsIdx++;
|
|
164
|
+
}
|
|
165
|
+
}
|
|
166
|
+
}
|
|
167
|
+
}
|
|
168
|
+
}
|
|
169
|
+
}
|
|
170
|
+
else {
|
|
171
|
+
// Chart with cell references (common in ODS spreadsheets)
|
|
172
|
+
// Extract series info from chart:series elements
|
|
173
|
+
const seriesNodes = (0, xmlUtils_1.getElementsByTagName)(chart, "chart:series");
|
|
174
|
+
for (const series of seriesNodes) {
|
|
175
|
+
// Series label is in chart:label-cell-address attribute
|
|
176
|
+
const labelAddr = series.getAttribute("chart:label-cell-address");
|
|
177
|
+
const valuesAddr = series.getAttribute("chart:values-cell-range-address");
|
|
178
|
+
// Try to get series name from any text:p in a title
|
|
179
|
+
let name = undefined;
|
|
180
|
+
const seriesLabels = (0, xmlUtils_1.getElementsByTagName)(series, "text:p");
|
|
181
|
+
if (seriesLabels.length > 0) {
|
|
182
|
+
name = seriesLabels[0].textContent || undefined;
|
|
183
|
+
}
|
|
184
|
+
// If no embedded name, use cell address as identifier
|
|
185
|
+
if (!name && labelAddr) {
|
|
186
|
+
// Extract just the cell reference part, e.g., "Sheet1.$D$2" -> "D2"
|
|
187
|
+
const cellRef = labelAddr.split('.').pop()?.replace(/\$/g, '') || labelAddr;
|
|
188
|
+
name = `Series ${cellRef}`;
|
|
189
|
+
}
|
|
190
|
+
// Values cell range can give us some info
|
|
191
|
+
const dataSet = {
|
|
192
|
+
name,
|
|
193
|
+
values: [],
|
|
194
|
+
pointLabels: []
|
|
195
|
+
};
|
|
196
|
+
// Add values address as a hint in rawTexts
|
|
197
|
+
if (valuesAddr) {
|
|
198
|
+
dataSet.values.push(`[${valuesAddr}]`);
|
|
199
|
+
}
|
|
200
|
+
dataSets.push(dataSet);
|
|
201
|
+
}
|
|
202
|
+
// Extract category labels from chart:categories
|
|
203
|
+
const categories = (0, xmlUtils_1.getElementsByTagName)(chart, "chart:categories")[0];
|
|
204
|
+
if (categories) {
|
|
205
|
+
const catRange = categories.getAttribute("table:cell-range-address");
|
|
206
|
+
if (catRange) {
|
|
207
|
+
labels.push(`[${catRange}]`);
|
|
208
|
+
}
|
|
209
|
+
}
|
|
210
|
+
}
|
|
211
|
+
// Axis titles
|
|
212
|
+
const axes = (0, xmlUtils_1.getElementsByTagName)(dom, "chart:axis");
|
|
213
|
+
let xAxisTitle = undefined;
|
|
214
|
+
let yAxisTitle = undefined;
|
|
215
|
+
for (const axis of axes) {
|
|
216
|
+
const dimension = axis.getAttribute("chart:dimension");
|
|
217
|
+
const axisTitleNode = (0, xmlUtils_1.getElementsByTagName)(axis, "chart:title")[0];
|
|
218
|
+
const axisTitle = axisTitleNode ? (0, xmlUtils_1.getElementsByTagName)(axisTitleNode, "text:p")[0]?.textContent || undefined : undefined;
|
|
219
|
+
if (dimension === 'x')
|
|
220
|
+
xAxisTitle = axisTitle;
|
|
221
|
+
else if (dimension === 'y')
|
|
222
|
+
yAxisTitle = axisTitle;
|
|
223
|
+
}
|
|
224
|
+
// Structured rawTexts: title + series info
|
|
225
|
+
if (title)
|
|
226
|
+
rawTexts.push(title);
|
|
227
|
+
for (const ds of dataSets) {
|
|
228
|
+
if (ds.name)
|
|
229
|
+
rawTexts.push(ds.name);
|
|
230
|
+
rawTexts.push(...labels);
|
|
231
|
+
rawTexts.push(...ds.values);
|
|
232
|
+
}
|
|
233
|
+
return {
|
|
234
|
+
title,
|
|
235
|
+
xAxisTitle,
|
|
236
|
+
yAxisTitle,
|
|
237
|
+
dataSets,
|
|
238
|
+
labels,
|
|
239
|
+
rawTexts
|
|
240
|
+
};
|
|
241
|
+
};
|
|
242
|
+
/**
|
|
243
|
+
* Universal chart data extractor that selects logic based on XML content.
|
|
244
|
+
* @param xmlBuffer Chart XML buffer
|
|
245
|
+
*/
|
|
246
|
+
const extractChartData = (xmlBuffer) => {
|
|
247
|
+
const head = xmlBuffer.toString("utf8", 0, 500);
|
|
248
|
+
if (head.includes("urn:oasis:names:tc:opendocument:xmlns:chart:1.0")) {
|
|
249
|
+
return extractOdfChartData(xmlBuffer);
|
|
250
|
+
}
|
|
251
|
+
else {
|
|
252
|
+
return extractOpenXmlChartData(xmlBuffer);
|
|
253
|
+
}
|
|
254
|
+
};
|
|
255
|
+
exports.extractChartData = extractChartData;
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Error Handling Utilities
|
|
3
|
+
*
|
|
4
|
+
* This module provides centralized error management for the OfficeParser library.
|
|
5
|
+
* It defines standard error types, messages, and handling logic to ensure
|
|
6
|
+
* consistent error reporting across all parsers and the main entry point.
|
|
7
|
+
*/
|
|
8
|
+
import { OfficeParserConfig } from '../types';
|
|
9
|
+
/**
|
|
10
|
+
* Standard error types for OfficeParser.
|
|
11
|
+
* Use these to identify the kind of error being reported.
|
|
12
|
+
*/
|
|
13
|
+
export declare enum OfficeErrorType {
|
|
14
|
+
/** Unsupported file extension */
|
|
15
|
+
EXTENSION_UNSUPPORTED = "EXTENSION_UNSUPPORTED",
|
|
16
|
+
/** File appears to be corrupted or malformed */
|
|
17
|
+
FILE_CORRUPTED = "FILE_CORRUPTED",
|
|
18
|
+
/** File could not be found at the specified path */
|
|
19
|
+
FILE_DOES_NOT_EXIST = "FILE_DOES_NOT_EXIST",
|
|
20
|
+
/** Specified location/directory is not reachable or is a directory */
|
|
21
|
+
LOCATION_NOT_FOUND = "LOCATION_NOT_FOUND",
|
|
22
|
+
/** Arguments passed to the function are missing or invalid */
|
|
23
|
+
IMPROPER_ARGUMENTS = "IMPROPER_ARGUMENTS",
|
|
24
|
+
/** Error occurred while reading or processing file buffers */
|
|
25
|
+
IMPROPER_BUFFERS = "IMPROPER_BUFFERS",
|
|
26
|
+
/** Input type is not a supported type (string, Buffer, ArrayBuffer) */
|
|
27
|
+
INVALID_INPUT = "INVALID_INPUT",
|
|
28
|
+
/** PDF worker source is missing (required in browser) */
|
|
29
|
+
PDF_WORKER_MISSING = "PDF_WORKER_MISSING"
|
|
30
|
+
}
|
|
31
|
+
/**
|
|
32
|
+
* Creates, optionally logs to console, and returns a formatted OfficeParser error.
|
|
33
|
+
*
|
|
34
|
+
* @param type - The type of error
|
|
35
|
+
* @param config - Parser configuration (checks outputErrorToConsole)
|
|
36
|
+
* @param info - Optional additional information
|
|
37
|
+
* @returns The Error object to be thrown
|
|
38
|
+
*/
|
|
39
|
+
export declare const getOfficeError: (type: OfficeErrorType, config: OfficeParserConfig, info?: any) => Error;
|
|
40
|
+
/**
|
|
41
|
+
* Wraps an existing error with OfficeParser context and performs corruption detection.
|
|
42
|
+
* Optionally logs the error to console.
|
|
43
|
+
*
|
|
44
|
+
* @param error - The original error object
|
|
45
|
+
* @param config - Parser configuration
|
|
46
|
+
* @param filePath - Optional file path for context
|
|
47
|
+
* @returns The wrapped Error object to be thrown
|
|
48
|
+
*/
|
|
49
|
+
export declare const getWrappedError: (error: any, config: OfficeParserConfig, filePath?: string) => Error;
|
|
50
|
+
/**
|
|
51
|
+
* Conditionally logs a warning message to the console.
|
|
52
|
+
* Used for non-fatal errors that shouldn't stop the parsing process.
|
|
53
|
+
*
|
|
54
|
+
* @param message - The warning message
|
|
55
|
+
* @param config - Parser configuration
|
|
56
|
+
* @param error - Optional original error object for more context
|
|
57
|
+
*/
|
|
58
|
+
export declare const logWarning: (message: string, config: OfficeParserConfig, error?: any) => void;
|
|
@@ -0,0 +1,120 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* Error Handling Utilities
|
|
4
|
+
*
|
|
5
|
+
* This module provides centralized error management for the OfficeParser library.
|
|
6
|
+
* It defines standard error types, messages, and handling logic to ensure
|
|
7
|
+
* consistent error reporting across all parsers and the main entry point.
|
|
8
|
+
*/
|
|
9
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
10
|
+
exports.logWarning = exports.getWrappedError = exports.getOfficeError = exports.OfficeErrorType = void 0;
|
|
11
|
+
/** Error header prefix for all error messages */
|
|
12
|
+
const ERRORHEADER = "[OfficeParser]: ";
|
|
13
|
+
/**
|
|
14
|
+
* Standard error types for OfficeParser.
|
|
15
|
+
* Use these to identify the kind of error being reported.
|
|
16
|
+
*/
|
|
17
|
+
var OfficeErrorType;
|
|
18
|
+
(function (OfficeErrorType) {
|
|
19
|
+
/** Unsupported file extension */
|
|
20
|
+
OfficeErrorType["EXTENSION_UNSUPPORTED"] = "EXTENSION_UNSUPPORTED";
|
|
21
|
+
/** File appears to be corrupted or malformed */
|
|
22
|
+
OfficeErrorType["FILE_CORRUPTED"] = "FILE_CORRUPTED";
|
|
23
|
+
/** File could not be found at the specified path */
|
|
24
|
+
OfficeErrorType["FILE_DOES_NOT_EXIST"] = "FILE_DOES_NOT_EXIST";
|
|
25
|
+
/** Specified location/directory is not reachable or is a directory */
|
|
26
|
+
OfficeErrorType["LOCATION_NOT_FOUND"] = "LOCATION_NOT_FOUND";
|
|
27
|
+
/** Arguments passed to the function are missing or invalid */
|
|
28
|
+
OfficeErrorType["IMPROPER_ARGUMENTS"] = "IMPROPER_ARGUMENTS";
|
|
29
|
+
/** Error occurred while reading or processing file buffers */
|
|
30
|
+
OfficeErrorType["IMPROPER_BUFFERS"] = "IMPROPER_BUFFERS";
|
|
31
|
+
/** Input type is not a supported type (string, Buffer, ArrayBuffer) */
|
|
32
|
+
OfficeErrorType["INVALID_INPUT"] = "INVALID_INPUT";
|
|
33
|
+
/** PDF worker source is missing (required in browser) */
|
|
34
|
+
OfficeErrorType["PDF_WORKER_MISSING"] = "PDF_WORKER_MISSING";
|
|
35
|
+
})(OfficeErrorType = exports.OfficeErrorType || (exports.OfficeErrorType = {}));
|
|
36
|
+
/**
|
|
37
|
+
* Lookup table for error messages.
|
|
38
|
+
* Some entries are functions that take parameters to build dynamic messages.
|
|
39
|
+
*/
|
|
40
|
+
const ERROR_MESSAGES = {
|
|
41
|
+
[OfficeErrorType.EXTENSION_UNSUPPORTED]: (ext) => `Sorry, OfficeParser currently supports docx, pptx, xlsx, odt, odp, ods, pdf, rtf files only. Create a ticket in Issues on github to add support for ${ext} files. Stay tuned for further updates.`,
|
|
42
|
+
[OfficeErrorType.FILE_CORRUPTED]: (filepath) => `Your file ${filepath} seems to be corrupted. If you are sure it is fine, please create a ticket in Issues on github with the file to reproduce error.`,
|
|
43
|
+
[OfficeErrorType.FILE_DOES_NOT_EXIST]: (filepath) => `File ${filepath} could not be found! Check if the file exists or verify if the relative path to the file is correct from your terminal's location.`,
|
|
44
|
+
[OfficeErrorType.LOCATION_NOT_FOUND]: (location) => `Entered location ${location} is not reachable! Please make sure that the entered directory location exists. Check relative paths and reenter.`,
|
|
45
|
+
[OfficeErrorType.IMPROPER_ARGUMENTS]: `Improper arguments`,
|
|
46
|
+
[OfficeErrorType.IMPROPER_BUFFERS]: `Error occured while reading the file buffers`,
|
|
47
|
+
[OfficeErrorType.INVALID_INPUT]: `Invalid input type: Expected a Buffer or a valid file path`,
|
|
48
|
+
[OfficeErrorType.PDF_WORKER_MISSING]: `Missing PDF worker configuration. PDF parsing in browser environments requires a worker source. Please provide "pdfWorkerSrc" in your configuration.`
|
|
49
|
+
};
|
|
50
|
+
/**
|
|
51
|
+
* Creates a formatted error message for a specific error type.
|
|
52
|
+
*
|
|
53
|
+
* @param type - The type of error
|
|
54
|
+
* @param info - Optional additional information (e.g., filepath, extension)
|
|
55
|
+
* @returns The formatted error message string
|
|
56
|
+
*/
|
|
57
|
+
const createOfficeError = (type, info) => {
|
|
58
|
+
const msg = ERROR_MESSAGES[type];
|
|
59
|
+
const message = typeof msg === 'function' ? msg(info) : msg;
|
|
60
|
+
return message;
|
|
61
|
+
};
|
|
62
|
+
/**
|
|
63
|
+
* Creates, optionally logs to console, and returns a formatted OfficeParser error.
|
|
64
|
+
*
|
|
65
|
+
* @param type - The type of error
|
|
66
|
+
* @param config - Parser configuration (checks outputErrorToConsole)
|
|
67
|
+
* @param info - Optional additional information
|
|
68
|
+
* @returns The Error object to be thrown
|
|
69
|
+
*/
|
|
70
|
+
const getOfficeError = (type, config, info) => {
|
|
71
|
+
const message = createOfficeError(type, info);
|
|
72
|
+
if (config.outputErrorToConsole) {
|
|
73
|
+
console.error(ERRORHEADER + message);
|
|
74
|
+
}
|
|
75
|
+
return new Error(ERRORHEADER + message);
|
|
76
|
+
};
|
|
77
|
+
exports.getOfficeError = getOfficeError;
|
|
78
|
+
/**
|
|
79
|
+
* Wraps an existing error with OfficeParser context and performs corruption detection.
|
|
80
|
+
* Optionally logs the error to console.
|
|
81
|
+
*
|
|
82
|
+
* @param error - The original error object
|
|
83
|
+
* @param config - Parser configuration
|
|
84
|
+
* @param filePath - Optional file path for context
|
|
85
|
+
* @returns The wrapped Error object to be thrown
|
|
86
|
+
*/
|
|
87
|
+
const getWrappedError = (error, config, filePath) => {
|
|
88
|
+
let message = error.message || error;
|
|
89
|
+
// Detect file corruption from common library error messages
|
|
90
|
+
if (filePath && (message.includes('end of central directory record') ||
|
|
91
|
+
message.includes('invalid XML') ||
|
|
92
|
+
message.includes('Failed to open zip file') ||
|
|
93
|
+
message.includes('invalid distance too far back'))) {
|
|
94
|
+
message = createOfficeError(OfficeErrorType.FILE_CORRUPTED, filePath);
|
|
95
|
+
}
|
|
96
|
+
if (config.outputErrorToConsole) {
|
|
97
|
+
console.error(ERRORHEADER + message);
|
|
98
|
+
}
|
|
99
|
+
return new Error(ERRORHEADER + message);
|
|
100
|
+
};
|
|
101
|
+
exports.getWrappedError = getWrappedError;
|
|
102
|
+
/**
|
|
103
|
+
* Conditionally logs a warning message to the console.
|
|
104
|
+
* Used for non-fatal errors that shouldn't stop the parsing process.
|
|
105
|
+
*
|
|
106
|
+
* @param message - The warning message
|
|
107
|
+
* @param config - Parser configuration
|
|
108
|
+
* @param error - Optional original error object for more context
|
|
109
|
+
*/
|
|
110
|
+
const logWarning = (message, config, error) => {
|
|
111
|
+
if (config.outputErrorToConsole) {
|
|
112
|
+
if (error) {
|
|
113
|
+
console.warn(ERRORHEADER + message, error);
|
|
114
|
+
}
|
|
115
|
+
else {
|
|
116
|
+
console.warn(ERRORHEADER + message);
|
|
117
|
+
}
|
|
118
|
+
}
|
|
119
|
+
};
|
|
120
|
+
exports.logWarning = logWarning;
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Image Processing Utilities
|
|
3
|
+
*
|
|
4
|
+
* Provides helper functions for working with image attachments extracted from office documents.
|
|
5
|
+
* Handles MIME type conversions, file extension mapping, and attachment object creation.
|
|
6
|
+
*
|
|
7
|
+
* @module imageUtils
|
|
8
|
+
*/
|
|
9
|
+
/// <reference types="node" />
|
|
10
|
+
import { OfficeAttachment } from '../types';
|
|
11
|
+
/**
|
|
12
|
+
* Converts a file extension to its corresponding MIME type.
|
|
13
|
+
*
|
|
14
|
+
* Used when creating attachments to determine the MIME type from a filename.
|
|
15
|
+
* The extension check is case-insensitive.
|
|
16
|
+
*
|
|
17
|
+
* @param ext - The file extension (with or without a dot, e.g., 'png', '.png')
|
|
18
|
+
* @returns The corresponding MIME type string
|
|
19
|
+
* @example
|
|
20
|
+
* ```typescript
|
|
21
|
+
* getMimeFromExtension('png'); // Returns 'image/png'
|
|
22
|
+
* getMimeFromExtension('JPG'); // Returns 'image/jpeg' (case-insensitive)
|
|
23
|
+
* getMimeFromExtension('unknown'); // Returns 'application/octet-stream'
|
|
24
|
+
* ```
|
|
25
|
+
*/
|
|
26
|
+
export declare const getMimeFromExtension: (ext: string) => string;
|
|
27
|
+
/**
|
|
28
|
+
* Detects the MIME type from file magic bytes (file signature).
|
|
29
|
+
*
|
|
30
|
+
* This is useful for files with incorrect or missing extensions (like .tmp files).
|
|
31
|
+
* Inspects the first few bytes of the file to determine the actual format.
|
|
32
|
+
*
|
|
33
|
+
* @param buffer - The file content as a Buffer
|
|
34
|
+
* @returns The detected MIME type, or undefined if not recognized
|
|
35
|
+
* @example
|
|
36
|
+
* ```typescript
|
|
37
|
+
* const pngBuffer = fs.readFileSync('image.tmp');
|
|
38
|
+
* getMimeFromBytes(pngBuffer); // Returns 'image/png' if it's a PNG file
|
|
39
|
+
* ```
|
|
40
|
+
*/
|
|
41
|
+
export declare const getMimeFromBytes: (buffer: Buffer) => string | undefined;
|
|
42
|
+
/**
|
|
43
|
+
* Creates an OfficeAttachment object from image data.
|
|
44
|
+
*
|
|
45
|
+
* This is a convenience function that:
|
|
46
|
+
* 1. Extracts the file extension from the filename
|
|
47
|
+
* 2. Determines the MIME type from the extension
|
|
48
|
+
* 3. Encodes the image buffer as Base64
|
|
49
|
+
* 4. Constructs a properly formatted OfficeAttachment object
|
|
50
|
+
*
|
|
51
|
+
* @param name - The filename of the image (e.g., 'image1.png', 'chart.jpg')
|
|
52
|
+
* @param content - The image data as a Node.js Buffer
|
|
53
|
+
* @returns An OfficeAttachment object ready to be added to the attachments array
|
|
54
|
+
*
|
|
55
|
+
* @example
|
|
56
|
+
* ```typescript
|
|
57
|
+
* const imageBuffer = fs.readFileSync('photo.png');
|
|
58
|
+
* const attachment = createAttachment('photo.png', imageBuffer);
|
|
59
|
+
*
|
|
60
|
+
* console.log(attachment.type); // 'image'
|
|
61
|
+
* console.log(attachment.mimeType); // 'image/png'
|
|
62
|
+
* console.log(attachment.name); // 'photo.png'
|
|
63
|
+
* console.log(attachment.extension); // 'png'
|
|
64
|
+
* console.log(attachment.data); // 'iVBORw0KGgoAAAANSUhEUgAA...' (Base64)
|
|
65
|
+
* ```
|
|
66
|
+
*/
|
|
67
|
+
export declare const createAttachment: (name: string, content: Buffer) => OfficeAttachment;
|
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* Image Processing Utilities
|
|
4
|
+
*
|
|
5
|
+
* Provides helper functions for working with image attachments extracted from office documents.
|
|
6
|
+
* Handles MIME type conversions, file extension mapping, and attachment object creation.
|
|
7
|
+
*
|
|
8
|
+
* @module imageUtils
|
|
9
|
+
*/
|
|
10
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
11
|
+
exports.createAttachment = exports.getMimeFromBytes = exports.getMimeFromExtension = void 0;
|
|
12
|
+
/**
|
|
13
|
+
* Converts a file extension to its corresponding MIME type.
|
|
14
|
+
*
|
|
15
|
+
* Used when creating attachments to determine the MIME type from a filename.
|
|
16
|
+
* The extension check is case-insensitive.
|
|
17
|
+
*
|
|
18
|
+
* @param ext - The file extension (with or without a dot, e.g., 'png', '.png')
|
|
19
|
+
* @returns The corresponding MIME type string
|
|
20
|
+
* @example
|
|
21
|
+
* ```typescript
|
|
22
|
+
* getMimeFromExtension('png'); // Returns 'image/png'
|
|
23
|
+
* getMimeFromExtension('JPG'); // Returns 'image/jpeg' (case-insensitive)
|
|
24
|
+
* getMimeFromExtension('unknown'); // Returns 'application/octet-stream'
|
|
25
|
+
* ```
|
|
26
|
+
*/
|
|
27
|
+
const getMimeFromExtension = (ext) => {
|
|
28
|
+
switch (ext.toLowerCase()) {
|
|
29
|
+
case 'jpg':
|
|
30
|
+
case 'jpeg': return 'image/jpeg';
|
|
31
|
+
case 'png': return 'image/png';
|
|
32
|
+
case 'gif': return 'image/gif';
|
|
33
|
+
case 'bmp': return 'image/bmp';
|
|
34
|
+
case 'tiff': return 'image/tiff';
|
|
35
|
+
case 'webp': return 'image/webp';
|
|
36
|
+
default: return 'application/octet-stream'; // Generic binary MIME type
|
|
37
|
+
}
|
|
38
|
+
};
|
|
39
|
+
exports.getMimeFromExtension = getMimeFromExtension;
|
|
40
|
+
/**
|
|
41
|
+
* Detects the MIME type from file magic bytes (file signature).
|
|
42
|
+
*
|
|
43
|
+
* This is useful for files with incorrect or missing extensions (like .tmp files).
|
|
44
|
+
* Inspects the first few bytes of the file to determine the actual format.
|
|
45
|
+
*
|
|
46
|
+
* @param buffer - The file content as a Buffer
|
|
47
|
+
* @returns The detected MIME type, or undefined if not recognized
|
|
48
|
+
* @example
|
|
49
|
+
* ```typescript
|
|
50
|
+
* const pngBuffer = fs.readFileSync('image.tmp');
|
|
51
|
+
* getMimeFromBytes(pngBuffer); // Returns 'image/png' if it's a PNG file
|
|
52
|
+
* ```
|
|
53
|
+
*/
|
|
54
|
+
const getMimeFromBytes = (buffer) => {
|
|
55
|
+
if (buffer.length < 4)
|
|
56
|
+
return undefined;
|
|
57
|
+
// PNG: 89 50 4E 47 (0x89 "PNG")
|
|
58
|
+
if (buffer[0] === 0x89 && buffer[1] === 0x50 && buffer[2] === 0x4E && buffer[3] === 0x47) {
|
|
59
|
+
return 'image/png';
|
|
60
|
+
}
|
|
61
|
+
// JPEG: FF D8 FF
|
|
62
|
+
if (buffer[0] === 0xFF && buffer[1] === 0xD8 && buffer[2] === 0xFF) {
|
|
63
|
+
return 'image/jpeg';
|
|
64
|
+
}
|
|
65
|
+
// GIF: 47 49 46 38 ("GIF8")
|
|
66
|
+
if (buffer[0] === 0x47 && buffer[1] === 0x49 && buffer[2] === 0x46 && buffer[3] === 0x38) {
|
|
67
|
+
return 'image/gif';
|
|
68
|
+
}
|
|
69
|
+
// BMP: 42 4D ("BM")
|
|
70
|
+
if (buffer[0] === 0x42 && buffer[1] === 0x4D) {
|
|
71
|
+
return 'image/bmp';
|
|
72
|
+
}
|
|
73
|
+
// TIFF: 49 49 2A 00 (little endian) or 4D 4D 00 2A (big endian)
|
|
74
|
+
if ((buffer[0] === 0x49 && buffer[1] === 0x49 && buffer[2] === 0x2A && buffer[3] === 0x00) ||
|
|
75
|
+
(buffer[0] === 0x4D && buffer[1] === 0x4D && buffer[2] === 0x00 && buffer[3] === 0x2A)) {
|
|
76
|
+
return 'image/tiff';
|
|
77
|
+
}
|
|
78
|
+
// WebP: 52 49 46 46 ... 57 45 42 50 ("RIFF" ... "WEBP")
|
|
79
|
+
if (buffer.length >= 12 &&
|
|
80
|
+
buffer[0] === 0x52 && buffer[1] === 0x49 && buffer[2] === 0x46 && buffer[3] === 0x46 &&
|
|
81
|
+
buffer[8] === 0x57 && buffer[9] === 0x45 && buffer[10] === 0x42 && buffer[11] === 0x50) {
|
|
82
|
+
return 'image/webp';
|
|
83
|
+
}
|
|
84
|
+
return undefined;
|
|
85
|
+
};
|
|
86
|
+
exports.getMimeFromBytes = getMimeFromBytes;
|
|
87
|
+
/**
|
|
88
|
+
* Creates an OfficeAttachment object from image data.
|
|
89
|
+
*
|
|
90
|
+
* This is a convenience function that:
|
|
91
|
+
* 1. Extracts the file extension from the filename
|
|
92
|
+
* 2. Determines the MIME type from the extension
|
|
93
|
+
* 3. Encodes the image buffer as Base64
|
|
94
|
+
* 4. Constructs a properly formatted OfficeAttachment object
|
|
95
|
+
*
|
|
96
|
+
* @param name - The filename of the image (e.g., 'image1.png', 'chart.jpg')
|
|
97
|
+
* @param content - The image data as a Node.js Buffer
|
|
98
|
+
* @returns An OfficeAttachment object ready to be added to the attachments array
|
|
99
|
+
*
|
|
100
|
+
* @example
|
|
101
|
+
* ```typescript
|
|
102
|
+
* const imageBuffer = fs.readFileSync('photo.png');
|
|
103
|
+
* const attachment = createAttachment('photo.png', imageBuffer);
|
|
104
|
+
*
|
|
105
|
+
* console.log(attachment.type); // 'image'
|
|
106
|
+
* console.log(attachment.mimeType); // 'image/png'
|
|
107
|
+
* console.log(attachment.name); // 'photo.png'
|
|
108
|
+
* console.log(attachment.extension); // 'png'
|
|
109
|
+
* console.log(attachment.data); // 'iVBORw0KGgoAAAANSUhEUgAA...' (Base64)
|
|
110
|
+
* ```
|
|
111
|
+
*/
|
|
112
|
+
const createAttachment = (name, content) => {
|
|
113
|
+
// Step 1: Extract file extension from the filename
|
|
114
|
+
// Example: 'image1.png' -> 'png'
|
|
115
|
+
const ext = name.split('.').pop() || '';
|
|
116
|
+
// Step 2: Try to detect MIME type from magic bytes first (more reliable)
|
|
117
|
+
// This handles cases like .tmp files or mismatched extensions
|
|
118
|
+
let mime = (0, exports.getMimeFromBytes)(content);
|
|
119
|
+
// Step 3: Fallback to extension if magic bytes couldn't detect it
|
|
120
|
+
// Useful for SVG or formats not covered by getMimeFromBytes
|
|
121
|
+
if (!mime) {
|
|
122
|
+
mime = (0, exports.getMimeFromExtension)(ext);
|
|
123
|
+
}
|
|
124
|
+
// Step 4: Create and return the attachment object
|
|
125
|
+
return {
|
|
126
|
+
type: 'image',
|
|
127
|
+
mimeType: mime,
|
|
128
|
+
data: content.toString('base64'),
|
|
129
|
+
name: name,
|
|
130
|
+
extension: ext // File extension for reference
|
|
131
|
+
};
|
|
132
|
+
};
|
|
133
|
+
exports.createAttachment = createAttachment;
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* OCR (Optical Character Recognition) Utilities
|
|
3
|
+
*
|
|
4
|
+
* This module provides functions for extracting text from images using Tesseract.js.
|
|
5
|
+
* Used when `config.ocr` is enabled to extract text from embedded images in documents.
|
|
6
|
+
*
|
|
7
|
+
* @module ocrUtils
|
|
8
|
+
*/
|
|
9
|
+
/// <reference types="node" />
|
|
10
|
+
/**
|
|
11
|
+
* Performs Optical Character Recognition (OCR) on an image to extract text.
|
|
12
|
+
*
|
|
13
|
+
* Uses Tesseract.js to recognize text in the provided image buffer.
|
|
14
|
+
* This is useful for extracting text from screenshots, scanned documents,
|
|
15
|
+
* charts with labels, or any image containing text.
|
|
16
|
+
*
|
|
17
|
+
* The function creates a new Tesseract worker, processes the image,
|
|
18
|
+
* and properly terminates the worker to free resources.
|
|
19
|
+
*
|
|
20
|
+
* @param imageBuffer - The image data as a Node.js Buffer (PNG, JPEG, etc.)
|
|
21
|
+
* @param language - The language code for OCR (default: 'eng' for English).
|
|
22
|
+
* Supports ISO 639-2/T three-letter codes: 'eng', 'spa', 'fra', 'deu', etc.
|
|
23
|
+
* Multiple languages can be combined with '+': 'eng+fra'
|
|
24
|
+
* @returns A promise that resolves to the recognized text as a string
|
|
25
|
+
* @throws {Error} If the image cannot be processed or Tesseract initialization fails
|
|
26
|
+
*
|
|
27
|
+
* @example
|
|
28
|
+
* ```typescript
|
|
29
|
+
* // Extract text from an English image
|
|
30
|
+
* const text = await performOcr(imageBuffer, 'eng');
|
|
31
|
+
* console.log(text); // "Annual Revenue: $1.2M"
|
|
32
|
+
*
|
|
33
|
+
* // Extract text from a multilingual image
|
|
34
|
+
* const text = await performOcr(imageBuffer, 'eng+spa');
|
|
35
|
+
* ```
|
|
36
|
+
*
|
|
37
|
+
* @see https://github.com/naptha/tesseract.js for supported languages and options
|
|
38
|
+
*/
|
|
39
|
+
export declare const performOcr: (image: Buffer | string, language?: string) => Promise<string>;
|