officeparser 5.2.2 → 6.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,255 @@
1
+ "use strict";
2
+ Object.defineProperty(exports, "__esModule", { value: true });
3
+ exports.extractChartData = void 0;
4
+ const xmlUtils_1 = require("./xmlUtils");
5
+ /**
6
+ * Extracts a single text element located at:
7
+ * c:title -> c:tx -> c:rich -> a:p -> a:r -> a:t
8
+ * OR c:tx -> c:strRef -> c:strCache -> c:pt -> c:v
9
+ * @param el Chart element
10
+ * @param tagName The tag to search for (e.g., "c:title", "c:tx")
11
+ */
12
+ const extractOpenXmlRichText = (el, tagName) => {
13
+ const target = (el.localName === tagName || el.tagName === tagName) ? el : el.getElementsByTagName(tagName)[0];
14
+ if (!target)
15
+ return undefined;
16
+ // 1. Try c:rich or a:p (standard rich text)
17
+ const richNodes = target.getElementsByTagName("c:rich");
18
+ const pNodes = target.getElementsByTagName("a:p");
19
+ const textContainers = richNodes.length > 0 ? Array.from(richNodes) : Array.from(pNodes);
20
+ if (textContainers.length > 0) {
21
+ let acc = "";
22
+ for (const container of textContainers) {
23
+ const tNodes = container.getElementsByTagName("a:t");
24
+ for (let i = 0; i < tNodes.length; i++) {
25
+ acc += (tNodes[i].textContent || "") + " ";
26
+ }
27
+ }
28
+ if (acc.trim())
29
+ return acc.trim();
30
+ }
31
+ // 2. Try c:v (cached values/strings)
32
+ const vNode = target.getElementsByTagName("c:v")[0];
33
+ if (vNode && vNode.textContent) {
34
+ return vNode.textContent.trim() || undefined;
35
+ }
36
+ return undefined;
37
+ };
38
+ /**
39
+ * Extracts fully structured chart data from OpenXML (PPTX, XLSX) chart XML.
40
+ * @param xmlBuffer Chart XML buffer
41
+ */
42
+ const extractOpenXmlChartData = (xmlBuffer) => {
43
+ const xml = xmlBuffer.toString("utf8");
44
+ const dom = (0, xmlUtils_1.parseXmlString)(xml);
45
+ const root = dom.documentElement;
46
+ const title = extractOpenXmlRichText(root, "c:title");
47
+ // Axis titles
48
+ let xAxisTitle = undefined;
49
+ let yAxisTitle = undefined;
50
+ const catAxes = root.getElementsByTagName("c:catAx");
51
+ if (catAxes.length > 0)
52
+ xAxisTitle = extractOpenXmlRichText(catAxes[0], "c:title");
53
+ const valAxes = root.getElementsByTagName("c:valAx");
54
+ if (valAxes.length > 0)
55
+ yAxisTitle = extractOpenXmlRichText(valAxes[0], "c:title");
56
+ // Extract Series (dataSets)
57
+ const seriesNodes = root.getElementsByTagName("c:ser");
58
+ const dataSets = [];
59
+ const sharedLabels = [];
60
+ for (let i = 0; i < seriesNodes.length; i++) {
61
+ const ser = seriesNodes[i];
62
+ // DataSet Name (c:tx)
63
+ const name = extractOpenXmlRichText(ser, "c:tx");
64
+ // Values (c:val)
65
+ const values = [];
66
+ const valNode = ser.getElementsByTagName("c:val")[0] || ser.getElementsByTagName("c:yVal")[0];
67
+ if (valNode) {
68
+ const vNodes = valNode.getElementsByTagName("c:v");
69
+ for (let j = 0; j < vNodes.length; j++) {
70
+ const v = vNodes[j].textContent?.trim();
71
+ if (v)
72
+ values.push(v);
73
+ }
74
+ }
75
+ // Point Labels (data labels)
76
+ const pointLabels = [];
77
+ const dLbls = ser.getElementsByTagName("c:dLbl");
78
+ for (let j = 0; j < dLbls.length; j++) {
79
+ const lbl = extractOpenXmlRichText(dLbls[j], "c:tx");
80
+ if (lbl)
81
+ pointLabels.push(lbl);
82
+ }
83
+ // Categories (labels) - c:cat or c:xVal
84
+ const catNode = ser.getElementsByTagName("c:cat")[0] || ser.getElementsByTagName("c:xVal")[0];
85
+ if (catNode) {
86
+ const vNodes = catNode.getElementsByTagName("c:v");
87
+ const localLabels = [];
88
+ for (let j = 0; j < vNodes.length; j++) {
89
+ const v = vNodes[j].textContent?.trim();
90
+ if (v)
91
+ localLabels.push(v);
92
+ }
93
+ if (localLabels.length > 0 && sharedLabels.length === 0) {
94
+ sharedLabels.push(...localLabels);
95
+ }
96
+ }
97
+ dataSets.push({ name, values, pointLabels });
98
+ }
99
+ // Structured rawTexts: for each dataset: Name -> Labels -> Values
100
+ const rawTexts = [];
101
+ for (const ds of dataSets) {
102
+ if (ds.name)
103
+ rawTexts.push(ds.name);
104
+ rawTexts.push(...sharedLabels);
105
+ rawTexts.push(...ds.values);
106
+ }
107
+ return {
108
+ title,
109
+ xAxisTitle,
110
+ yAxisTitle,
111
+ dataSets,
112
+ labels: sharedLabels,
113
+ rawTexts
114
+ };
115
+ };
116
+ /**
117
+ * Extracts structured chart data from ODF (ODP, ODS) chart content.xml.
118
+ * @param xmlBuffer Chart XML buffer
119
+ */
120
+ const extractOdfChartData = (xmlBuffer) => {
121
+ const xml = xmlBuffer.toString("utf8");
122
+ const dom = (0, xmlUtils_1.parseXmlString)(xml);
123
+ const chart = (0, xmlUtils_1.getElementsByTagName)(dom, "chart:chart")[0] || dom.documentElement;
124
+ const titleNode = (0, xmlUtils_1.getElementsByTagName)(chart, "chart:title")[0];
125
+ const title = titleNode ? (0, xmlUtils_1.getElementsByTagName)(titleNode, "text:p")[0]?.textContent || undefined : undefined;
126
+ const table = (0, xmlUtils_1.getElementsByTagName)(chart, "table:table")[0];
127
+ const dataSets = [];
128
+ const labels = [];
129
+ const rawTexts = [];
130
+ if (table) {
131
+ // Chart with embedded data table (common in ODP presentations)
132
+ let rows = [];
133
+ const headerRowsNode = (0, xmlUtils_1.getDirectChildren)(table, "table:table-header-rows")[0];
134
+ if (headerRowsNode) {
135
+ rows.push(...(0, xmlUtils_1.getDirectChildren)(headerRowsNode, "table:table-row"));
136
+ }
137
+ rows.push(...(0, xmlUtils_1.getDirectChildren)(table, "table:table-row"));
138
+ if (rows.length > 0) {
139
+ // Header row for series names
140
+ const headerCells = (0, xmlUtils_1.getDirectChildren)(rows[0], "table:table-cell");
141
+ for (let j = 1; j < headerCells.length; j++) {
142
+ const colsRepeated = parseInt(headerCells[j].getAttribute("table:number-columns-repeated") || "1");
143
+ const name = (0, xmlUtils_1.getDirectChildren)(headerCells[j], "text:p")[0]?.textContent || undefined;
144
+ for (let k = 0; k < colsRepeated; k++) {
145
+ dataSets.push({ name, values: [], pointLabels: [] });
146
+ }
147
+ }
148
+ // Data rows
149
+ for (let i = 1; i < rows.length; i++) {
150
+ const dataCells = (0, xmlUtils_1.getDirectChildren)(rows[i], "table:table-cell");
151
+ if (dataCells.length > 0) {
152
+ const label = (0, xmlUtils_1.getDirectChildren)(dataCells[0], "text:p")[0]?.textContent || undefined;
153
+ if (label)
154
+ labels.push(label);
155
+ let dsIdx = 0;
156
+ for (let j = 1; j < dataCells.length; j++) {
157
+ const colsRepeated = parseInt(dataCells[j].getAttribute("table:number-columns-repeated") || "1");
158
+ const val = dataCells[j].getAttribute("office:value") || (0, xmlUtils_1.getDirectChildren)(dataCells[j], "text:p")[0]?.textContent || "";
159
+ for (let k = 0; k < colsRepeated; k++) {
160
+ if (dataSets[dsIdx]) {
161
+ dataSets[dsIdx].values.push(val);
162
+ }
163
+ dsIdx++;
164
+ }
165
+ }
166
+ }
167
+ }
168
+ }
169
+ }
170
+ else {
171
+ // Chart with cell references (common in ODS spreadsheets)
172
+ // Extract series info from chart:series elements
173
+ const seriesNodes = (0, xmlUtils_1.getElementsByTagName)(chart, "chart:series");
174
+ for (const series of seriesNodes) {
175
+ // Series label is in chart:label-cell-address attribute
176
+ const labelAddr = series.getAttribute("chart:label-cell-address");
177
+ const valuesAddr = series.getAttribute("chart:values-cell-range-address");
178
+ // Try to get series name from any text:p in a title
179
+ let name = undefined;
180
+ const seriesLabels = (0, xmlUtils_1.getElementsByTagName)(series, "text:p");
181
+ if (seriesLabels.length > 0) {
182
+ name = seriesLabels[0].textContent || undefined;
183
+ }
184
+ // If no embedded name, use cell address as identifier
185
+ if (!name && labelAddr) {
186
+ // Extract just the cell reference part, e.g., "Sheet1.$D$2" -> "D2"
187
+ const cellRef = labelAddr.split('.').pop()?.replace(/\$/g, '') || labelAddr;
188
+ name = `Series ${cellRef}`;
189
+ }
190
+ // Values cell range can give us some info
191
+ const dataSet = {
192
+ name,
193
+ values: [],
194
+ pointLabels: []
195
+ };
196
+ // Add values address as a hint in rawTexts
197
+ if (valuesAddr) {
198
+ dataSet.values.push(`[${valuesAddr}]`);
199
+ }
200
+ dataSets.push(dataSet);
201
+ }
202
+ // Extract category labels from chart:categories
203
+ const categories = (0, xmlUtils_1.getElementsByTagName)(chart, "chart:categories")[0];
204
+ if (categories) {
205
+ const catRange = categories.getAttribute("table:cell-range-address");
206
+ if (catRange) {
207
+ labels.push(`[${catRange}]`);
208
+ }
209
+ }
210
+ }
211
+ // Axis titles
212
+ const axes = (0, xmlUtils_1.getElementsByTagName)(dom, "chart:axis");
213
+ let xAxisTitle = undefined;
214
+ let yAxisTitle = undefined;
215
+ for (const axis of axes) {
216
+ const dimension = axis.getAttribute("chart:dimension");
217
+ const axisTitleNode = (0, xmlUtils_1.getElementsByTagName)(axis, "chart:title")[0];
218
+ const axisTitle = axisTitleNode ? (0, xmlUtils_1.getElementsByTagName)(axisTitleNode, "text:p")[0]?.textContent || undefined : undefined;
219
+ if (dimension === 'x')
220
+ xAxisTitle = axisTitle;
221
+ else if (dimension === 'y')
222
+ yAxisTitle = axisTitle;
223
+ }
224
+ // Structured rawTexts: title + series info
225
+ if (title)
226
+ rawTexts.push(title);
227
+ for (const ds of dataSets) {
228
+ if (ds.name)
229
+ rawTexts.push(ds.name);
230
+ rawTexts.push(...labels);
231
+ rawTexts.push(...ds.values);
232
+ }
233
+ return {
234
+ title,
235
+ xAxisTitle,
236
+ yAxisTitle,
237
+ dataSets,
238
+ labels,
239
+ rawTexts
240
+ };
241
+ };
242
+ /**
243
+ * Universal chart data extractor that selects logic based on XML content.
244
+ * @param xmlBuffer Chart XML buffer
245
+ */
246
+ const extractChartData = (xmlBuffer) => {
247
+ const head = xmlBuffer.toString("utf8", 0, 500);
248
+ if (head.includes("urn:oasis:names:tc:opendocument:xmlns:chart:1.0")) {
249
+ return extractOdfChartData(xmlBuffer);
250
+ }
251
+ else {
252
+ return extractOpenXmlChartData(xmlBuffer);
253
+ }
254
+ };
255
+ exports.extractChartData = extractChartData;
@@ -0,0 +1,58 @@
1
+ /**
2
+ * Error Handling Utilities
3
+ *
4
+ * This module provides centralized error management for the OfficeParser library.
5
+ * It defines standard error types, messages, and handling logic to ensure
6
+ * consistent error reporting across all parsers and the main entry point.
7
+ */
8
+ import { OfficeParserConfig } from '../types';
9
+ /**
10
+ * Standard error types for OfficeParser.
11
+ * Use these to identify the kind of error being reported.
12
+ */
13
+ export declare enum OfficeErrorType {
14
+ /** Unsupported file extension */
15
+ EXTENSION_UNSUPPORTED = "EXTENSION_UNSUPPORTED",
16
+ /** File appears to be corrupted or malformed */
17
+ FILE_CORRUPTED = "FILE_CORRUPTED",
18
+ /** File could not be found at the specified path */
19
+ FILE_DOES_NOT_EXIST = "FILE_DOES_NOT_EXIST",
20
+ /** Specified location/directory is not reachable or is a directory */
21
+ LOCATION_NOT_FOUND = "LOCATION_NOT_FOUND",
22
+ /** Arguments passed to the function are missing or invalid */
23
+ IMPROPER_ARGUMENTS = "IMPROPER_ARGUMENTS",
24
+ /** Error occurred while reading or processing file buffers */
25
+ IMPROPER_BUFFERS = "IMPROPER_BUFFERS",
26
+ /** Input type is not a supported type (string, Buffer, ArrayBuffer) */
27
+ INVALID_INPUT = "INVALID_INPUT",
28
+ /** PDF worker source is missing (required in browser) */
29
+ PDF_WORKER_MISSING = "PDF_WORKER_MISSING"
30
+ }
31
+ /**
32
+ * Creates, optionally logs to console, and returns a formatted OfficeParser error.
33
+ *
34
+ * @param type - The type of error
35
+ * @param config - Parser configuration (checks outputErrorToConsole)
36
+ * @param info - Optional additional information
37
+ * @returns The Error object to be thrown
38
+ */
39
+ export declare const getOfficeError: (type: OfficeErrorType, config: OfficeParserConfig, info?: any) => Error;
40
+ /**
41
+ * Wraps an existing error with OfficeParser context and performs corruption detection.
42
+ * Optionally logs the error to console.
43
+ *
44
+ * @param error - The original error object
45
+ * @param config - Parser configuration
46
+ * @param filePath - Optional file path for context
47
+ * @returns The wrapped Error object to be thrown
48
+ */
49
+ export declare const getWrappedError: (error: any, config: OfficeParserConfig, filePath?: string) => Error;
50
+ /**
51
+ * Conditionally logs a warning message to the console.
52
+ * Used for non-fatal errors that shouldn't stop the parsing process.
53
+ *
54
+ * @param message - The warning message
55
+ * @param config - Parser configuration
56
+ * @param error - Optional original error object for more context
57
+ */
58
+ export declare const logWarning: (message: string, config: OfficeParserConfig, error?: any) => void;
@@ -0,0 +1,120 @@
1
+ "use strict";
2
+ /**
3
+ * Error Handling Utilities
4
+ *
5
+ * This module provides centralized error management for the OfficeParser library.
6
+ * It defines standard error types, messages, and handling logic to ensure
7
+ * consistent error reporting across all parsers and the main entry point.
8
+ */
9
+ Object.defineProperty(exports, "__esModule", { value: true });
10
+ exports.logWarning = exports.getWrappedError = exports.getOfficeError = exports.OfficeErrorType = void 0;
11
+ /** Error header prefix for all error messages */
12
+ const ERRORHEADER = "[OfficeParser]: ";
13
+ /**
14
+ * Standard error types for OfficeParser.
15
+ * Use these to identify the kind of error being reported.
16
+ */
17
+ var OfficeErrorType;
18
+ (function (OfficeErrorType) {
19
+ /** Unsupported file extension */
20
+ OfficeErrorType["EXTENSION_UNSUPPORTED"] = "EXTENSION_UNSUPPORTED";
21
+ /** File appears to be corrupted or malformed */
22
+ OfficeErrorType["FILE_CORRUPTED"] = "FILE_CORRUPTED";
23
+ /** File could not be found at the specified path */
24
+ OfficeErrorType["FILE_DOES_NOT_EXIST"] = "FILE_DOES_NOT_EXIST";
25
+ /** Specified location/directory is not reachable or is a directory */
26
+ OfficeErrorType["LOCATION_NOT_FOUND"] = "LOCATION_NOT_FOUND";
27
+ /** Arguments passed to the function are missing or invalid */
28
+ OfficeErrorType["IMPROPER_ARGUMENTS"] = "IMPROPER_ARGUMENTS";
29
+ /** Error occurred while reading or processing file buffers */
30
+ OfficeErrorType["IMPROPER_BUFFERS"] = "IMPROPER_BUFFERS";
31
+ /** Input type is not a supported type (string, Buffer, ArrayBuffer) */
32
+ OfficeErrorType["INVALID_INPUT"] = "INVALID_INPUT";
33
+ /** PDF worker source is missing (required in browser) */
34
+ OfficeErrorType["PDF_WORKER_MISSING"] = "PDF_WORKER_MISSING";
35
+ })(OfficeErrorType = exports.OfficeErrorType || (exports.OfficeErrorType = {}));
36
+ /**
37
+ * Lookup table for error messages.
38
+ * Some entries are functions that take parameters to build dynamic messages.
39
+ */
40
+ const ERROR_MESSAGES = {
41
+ [OfficeErrorType.EXTENSION_UNSUPPORTED]: (ext) => `Sorry, OfficeParser currently supports docx, pptx, xlsx, odt, odp, ods, pdf, rtf files only. Create a ticket in Issues on github to add support for ${ext} files. Stay tuned for further updates.`,
42
+ [OfficeErrorType.FILE_CORRUPTED]: (filepath) => `Your file ${filepath} seems to be corrupted. If you are sure it is fine, please create a ticket in Issues on github with the file to reproduce error.`,
43
+ [OfficeErrorType.FILE_DOES_NOT_EXIST]: (filepath) => `File ${filepath} could not be found! Check if the file exists or verify if the relative path to the file is correct from your terminal's location.`,
44
+ [OfficeErrorType.LOCATION_NOT_FOUND]: (location) => `Entered location ${location} is not reachable! Please make sure that the entered directory location exists. Check relative paths and reenter.`,
45
+ [OfficeErrorType.IMPROPER_ARGUMENTS]: `Improper arguments`,
46
+ [OfficeErrorType.IMPROPER_BUFFERS]: `Error occured while reading the file buffers`,
47
+ [OfficeErrorType.INVALID_INPUT]: `Invalid input type: Expected a Buffer or a valid file path`,
48
+ [OfficeErrorType.PDF_WORKER_MISSING]: `Missing PDF worker configuration. PDF parsing in browser environments requires a worker source. Please provide "pdfWorkerSrc" in your configuration.`
49
+ };
50
+ /**
51
+ * Creates a formatted error message for a specific error type.
52
+ *
53
+ * @param type - The type of error
54
+ * @param info - Optional additional information (e.g., filepath, extension)
55
+ * @returns The formatted error message string
56
+ */
57
+ const createOfficeError = (type, info) => {
58
+ const msg = ERROR_MESSAGES[type];
59
+ const message = typeof msg === 'function' ? msg(info) : msg;
60
+ return message;
61
+ };
62
+ /**
63
+ * Creates, optionally logs to console, and returns a formatted OfficeParser error.
64
+ *
65
+ * @param type - The type of error
66
+ * @param config - Parser configuration (checks outputErrorToConsole)
67
+ * @param info - Optional additional information
68
+ * @returns The Error object to be thrown
69
+ */
70
+ const getOfficeError = (type, config, info) => {
71
+ const message = createOfficeError(type, info);
72
+ if (config.outputErrorToConsole) {
73
+ console.error(ERRORHEADER + message);
74
+ }
75
+ return new Error(ERRORHEADER + message);
76
+ };
77
+ exports.getOfficeError = getOfficeError;
78
+ /**
79
+ * Wraps an existing error with OfficeParser context and performs corruption detection.
80
+ * Optionally logs the error to console.
81
+ *
82
+ * @param error - The original error object
83
+ * @param config - Parser configuration
84
+ * @param filePath - Optional file path for context
85
+ * @returns The wrapped Error object to be thrown
86
+ */
87
+ const getWrappedError = (error, config, filePath) => {
88
+ let message = error.message || error;
89
+ // Detect file corruption from common library error messages
90
+ if (filePath && (message.includes('end of central directory record') ||
91
+ message.includes('invalid XML') ||
92
+ message.includes('Failed to open zip file') ||
93
+ message.includes('invalid distance too far back'))) {
94
+ message = createOfficeError(OfficeErrorType.FILE_CORRUPTED, filePath);
95
+ }
96
+ if (config.outputErrorToConsole) {
97
+ console.error(ERRORHEADER + message);
98
+ }
99
+ return new Error(ERRORHEADER + message);
100
+ };
101
+ exports.getWrappedError = getWrappedError;
102
+ /**
103
+ * Conditionally logs a warning message to the console.
104
+ * Used for non-fatal errors that shouldn't stop the parsing process.
105
+ *
106
+ * @param message - The warning message
107
+ * @param config - Parser configuration
108
+ * @param error - Optional original error object for more context
109
+ */
110
+ const logWarning = (message, config, error) => {
111
+ if (config.outputErrorToConsole) {
112
+ if (error) {
113
+ console.warn(ERRORHEADER + message, error);
114
+ }
115
+ else {
116
+ console.warn(ERRORHEADER + message);
117
+ }
118
+ }
119
+ };
120
+ exports.logWarning = logWarning;
@@ -0,0 +1,67 @@
1
+ /**
2
+ * Image Processing Utilities
3
+ *
4
+ * Provides helper functions for working with image attachments extracted from office documents.
5
+ * Handles MIME type conversions, file extension mapping, and attachment object creation.
6
+ *
7
+ * @module imageUtils
8
+ */
9
+ /// <reference types="node" />
10
+ import { OfficeAttachment } from '../types';
11
+ /**
12
+ * Converts a file extension to its corresponding MIME type.
13
+ *
14
+ * Used when creating attachments to determine the MIME type from a filename.
15
+ * The extension check is case-insensitive.
16
+ *
17
+ * @param ext - The file extension (with or without a dot, e.g., 'png', '.png')
18
+ * @returns The corresponding MIME type string
19
+ * @example
20
+ * ```typescript
21
+ * getMimeFromExtension('png'); // Returns 'image/png'
22
+ * getMimeFromExtension('JPG'); // Returns 'image/jpeg' (case-insensitive)
23
+ * getMimeFromExtension('unknown'); // Returns 'application/octet-stream'
24
+ * ```
25
+ */
26
+ export declare const getMimeFromExtension: (ext: string) => string;
27
+ /**
28
+ * Detects the MIME type from file magic bytes (file signature).
29
+ *
30
+ * This is useful for files with incorrect or missing extensions (like .tmp files).
31
+ * Inspects the first few bytes of the file to determine the actual format.
32
+ *
33
+ * @param buffer - The file content as a Buffer
34
+ * @returns The detected MIME type, or undefined if not recognized
35
+ * @example
36
+ * ```typescript
37
+ * const pngBuffer = fs.readFileSync('image.tmp');
38
+ * getMimeFromBytes(pngBuffer); // Returns 'image/png' if it's a PNG file
39
+ * ```
40
+ */
41
+ export declare const getMimeFromBytes: (buffer: Buffer) => string | undefined;
42
+ /**
43
+ * Creates an OfficeAttachment object from image data.
44
+ *
45
+ * This is a convenience function that:
46
+ * 1. Extracts the file extension from the filename
47
+ * 2. Determines the MIME type from the extension
48
+ * 3. Encodes the image buffer as Base64
49
+ * 4. Constructs a properly formatted OfficeAttachment object
50
+ *
51
+ * @param name - The filename of the image (e.g., 'image1.png', 'chart.jpg')
52
+ * @param content - The image data as a Node.js Buffer
53
+ * @returns An OfficeAttachment object ready to be added to the attachments array
54
+ *
55
+ * @example
56
+ * ```typescript
57
+ * const imageBuffer = fs.readFileSync('photo.png');
58
+ * const attachment = createAttachment('photo.png', imageBuffer);
59
+ *
60
+ * console.log(attachment.type); // 'image'
61
+ * console.log(attachment.mimeType); // 'image/png'
62
+ * console.log(attachment.name); // 'photo.png'
63
+ * console.log(attachment.extension); // 'png'
64
+ * console.log(attachment.data); // 'iVBORw0KGgoAAAANSUhEUgAA...' (Base64)
65
+ * ```
66
+ */
67
+ export declare const createAttachment: (name: string, content: Buffer) => OfficeAttachment;
@@ -0,0 +1,133 @@
1
+ "use strict";
2
+ /**
3
+ * Image Processing Utilities
4
+ *
5
+ * Provides helper functions for working with image attachments extracted from office documents.
6
+ * Handles MIME type conversions, file extension mapping, and attachment object creation.
7
+ *
8
+ * @module imageUtils
9
+ */
10
+ Object.defineProperty(exports, "__esModule", { value: true });
11
+ exports.createAttachment = exports.getMimeFromBytes = exports.getMimeFromExtension = void 0;
12
+ /**
13
+ * Converts a file extension to its corresponding MIME type.
14
+ *
15
+ * Used when creating attachments to determine the MIME type from a filename.
16
+ * The extension check is case-insensitive.
17
+ *
18
+ * @param ext - The file extension (with or without a dot, e.g., 'png', '.png')
19
+ * @returns The corresponding MIME type string
20
+ * @example
21
+ * ```typescript
22
+ * getMimeFromExtension('png'); // Returns 'image/png'
23
+ * getMimeFromExtension('JPG'); // Returns 'image/jpeg' (case-insensitive)
24
+ * getMimeFromExtension('unknown'); // Returns 'application/octet-stream'
25
+ * ```
26
+ */
27
+ const getMimeFromExtension = (ext) => {
28
+ switch (ext.toLowerCase()) {
29
+ case 'jpg':
30
+ case 'jpeg': return 'image/jpeg';
31
+ case 'png': return 'image/png';
32
+ case 'gif': return 'image/gif';
33
+ case 'bmp': return 'image/bmp';
34
+ case 'tiff': return 'image/tiff';
35
+ case 'webp': return 'image/webp';
36
+ default: return 'application/octet-stream'; // Generic binary MIME type
37
+ }
38
+ };
39
+ exports.getMimeFromExtension = getMimeFromExtension;
40
+ /**
41
+ * Detects the MIME type from file magic bytes (file signature).
42
+ *
43
+ * This is useful for files with incorrect or missing extensions (like .tmp files).
44
+ * Inspects the first few bytes of the file to determine the actual format.
45
+ *
46
+ * @param buffer - The file content as a Buffer
47
+ * @returns The detected MIME type, or undefined if not recognized
48
+ * @example
49
+ * ```typescript
50
+ * const pngBuffer = fs.readFileSync('image.tmp');
51
+ * getMimeFromBytes(pngBuffer); // Returns 'image/png' if it's a PNG file
52
+ * ```
53
+ */
54
+ const getMimeFromBytes = (buffer) => {
55
+ if (buffer.length < 4)
56
+ return undefined;
57
+ // PNG: 89 50 4E 47 (0x89 "PNG")
58
+ if (buffer[0] === 0x89 && buffer[1] === 0x50 && buffer[2] === 0x4E && buffer[3] === 0x47) {
59
+ return 'image/png';
60
+ }
61
+ // JPEG: FF D8 FF
62
+ if (buffer[0] === 0xFF && buffer[1] === 0xD8 && buffer[2] === 0xFF) {
63
+ return 'image/jpeg';
64
+ }
65
+ // GIF: 47 49 46 38 ("GIF8")
66
+ if (buffer[0] === 0x47 && buffer[1] === 0x49 && buffer[2] === 0x46 && buffer[3] === 0x38) {
67
+ return 'image/gif';
68
+ }
69
+ // BMP: 42 4D ("BM")
70
+ if (buffer[0] === 0x42 && buffer[1] === 0x4D) {
71
+ return 'image/bmp';
72
+ }
73
+ // TIFF: 49 49 2A 00 (little endian) or 4D 4D 00 2A (big endian)
74
+ if ((buffer[0] === 0x49 && buffer[1] === 0x49 && buffer[2] === 0x2A && buffer[3] === 0x00) ||
75
+ (buffer[0] === 0x4D && buffer[1] === 0x4D && buffer[2] === 0x00 && buffer[3] === 0x2A)) {
76
+ return 'image/tiff';
77
+ }
78
+ // WebP: 52 49 46 46 ... 57 45 42 50 ("RIFF" ... "WEBP")
79
+ if (buffer.length >= 12 &&
80
+ buffer[0] === 0x52 && buffer[1] === 0x49 && buffer[2] === 0x46 && buffer[3] === 0x46 &&
81
+ buffer[8] === 0x57 && buffer[9] === 0x45 && buffer[10] === 0x42 && buffer[11] === 0x50) {
82
+ return 'image/webp';
83
+ }
84
+ return undefined;
85
+ };
86
+ exports.getMimeFromBytes = getMimeFromBytes;
87
+ /**
88
+ * Creates an OfficeAttachment object from image data.
89
+ *
90
+ * This is a convenience function that:
91
+ * 1. Extracts the file extension from the filename
92
+ * 2. Determines the MIME type from the extension
93
+ * 3. Encodes the image buffer as Base64
94
+ * 4. Constructs a properly formatted OfficeAttachment object
95
+ *
96
+ * @param name - The filename of the image (e.g., 'image1.png', 'chart.jpg')
97
+ * @param content - The image data as a Node.js Buffer
98
+ * @returns An OfficeAttachment object ready to be added to the attachments array
99
+ *
100
+ * @example
101
+ * ```typescript
102
+ * const imageBuffer = fs.readFileSync('photo.png');
103
+ * const attachment = createAttachment('photo.png', imageBuffer);
104
+ *
105
+ * console.log(attachment.type); // 'image'
106
+ * console.log(attachment.mimeType); // 'image/png'
107
+ * console.log(attachment.name); // 'photo.png'
108
+ * console.log(attachment.extension); // 'png'
109
+ * console.log(attachment.data); // 'iVBORw0KGgoAAAANSUhEUgAA...' (Base64)
110
+ * ```
111
+ */
112
+ const createAttachment = (name, content) => {
113
+ // Step 1: Extract file extension from the filename
114
+ // Example: 'image1.png' -> 'png'
115
+ const ext = name.split('.').pop() || '';
116
+ // Step 2: Try to detect MIME type from magic bytes first (more reliable)
117
+ // This handles cases like .tmp files or mismatched extensions
118
+ let mime = (0, exports.getMimeFromBytes)(content);
119
+ // Step 3: Fallback to extension if magic bytes couldn't detect it
120
+ // Useful for SVG or formats not covered by getMimeFromBytes
121
+ if (!mime) {
122
+ mime = (0, exports.getMimeFromExtension)(ext);
123
+ }
124
+ // Step 4: Create and return the attachment object
125
+ return {
126
+ type: 'image',
127
+ mimeType: mime,
128
+ data: content.toString('base64'),
129
+ name: name,
130
+ extension: ext // File extension for reference
131
+ };
132
+ };
133
+ exports.createAttachment = createAttachment;
@@ -0,0 +1,39 @@
1
+ /**
2
+ * OCR (Optical Character Recognition) Utilities
3
+ *
4
+ * This module provides functions for extracting text from images using Tesseract.js.
5
+ * Used when `config.ocr` is enabled to extract text from embedded images in documents.
6
+ *
7
+ * @module ocrUtils
8
+ */
9
+ /// <reference types="node" />
10
+ /**
11
+ * Performs Optical Character Recognition (OCR) on an image to extract text.
12
+ *
13
+ * Uses Tesseract.js to recognize text in the provided image buffer.
14
+ * This is useful for extracting text from screenshots, scanned documents,
15
+ * charts with labels, or any image containing text.
16
+ *
17
+ * The function creates a new Tesseract worker, processes the image,
18
+ * and properly terminates the worker to free resources.
19
+ *
20
+ * @param imageBuffer - The image data as a Node.js Buffer (PNG, JPEG, etc.)
21
+ * @param language - The language code for OCR (default: 'eng' for English).
22
+ * Supports ISO 639-2/T three-letter codes: 'eng', 'spa', 'fra', 'deu', etc.
23
+ * Multiple languages can be combined with '+': 'eng+fra'
24
+ * @returns A promise that resolves to the recognized text as a string
25
+ * @throws {Error} If the image cannot be processed or Tesseract initialization fails
26
+ *
27
+ * @example
28
+ * ```typescript
29
+ * // Extract text from an English image
30
+ * const text = await performOcr(imageBuffer, 'eng');
31
+ * console.log(text); // "Annual Revenue: $1.2M"
32
+ *
33
+ * // Extract text from a multilingual image
34
+ * const text = await performOcr(imageBuffer, 'eng+spa');
35
+ * ```
36
+ *
37
+ * @see https://github.com/naptha/tesseract.js for supported languages and options
38
+ */
39
+ export declare const performOcr: (image: Buffer | string, language?: string) => Promise<string>;