officeparser 5.2.2 → 6.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,33 @@
1
+ /**
2
+ * Excel Spreadsheet (XLSX) Parser
3
+ *
4
+ * **XLSX Format Overview:**
5
+ * XLSX is the default format for Microsoft Excel since Office 2007, based on OOXML.
6
+ *
7
+ * **File Structure:**
8
+ * - `xl/workbook.xml` - Workbook structure and sheet list
9
+ * - `xl/worksheets/sheet1.xml` - Individual sheet data
10
+ * - `xl/sharedStrings.xml` - Shared string table (cell text)
11
+ * - `xl/styles.xml` - Cell styling information
12
+ * - `xl/drawings/*` - Charts and drawings
13
+ * - `xl/media/*` - Embedded images
14
+ *
15
+ * **Key Elements:**
16
+ * - `<row>` - Table row with row index
17
+ * - `<c r="A1">` - Cell with reference (A1, B2, etc.)
18
+ * - `<v>` - Cell value (number or shared string index)
19
+ * - `<t="s">` - Cell type (s=string, n=number, b=boolean)
20
+ *
21
+ * @module ExcelParser
22
+ * @see https://www.ecma-international.org/publications-and-standards/standards/ecma-376/
23
+ */
24
+ /// <reference types="node" />
25
+ import { OfficeParserAST, OfficeParserConfig } from '../types';
26
+ /**
27
+ * Parses an Excel spreadsheet (.xlsx) and extracts sheets, rows, and cells.
28
+ *
29
+ * @param buffer - The XLSX file as a Buffer
30
+ * @param config - Parser configuration
31
+ * @returns A promise resolving to the parsed AST
32
+ */
33
+ export declare const parseExcel: (buffer: Buffer, config: OfficeParserConfig) => Promise<OfficeParserAST>;
@@ -0,0 +1,643 @@
1
+ "use strict";
2
+ /**
3
+ * Excel Spreadsheet (XLSX) Parser
4
+ *
5
+ * **XLSX Format Overview:**
6
+ * XLSX is the default format for Microsoft Excel since Office 2007, based on OOXML.
7
+ *
8
+ * **File Structure:**
9
+ * - `xl/workbook.xml` - Workbook structure and sheet list
10
+ * - `xl/worksheets/sheet1.xml` - Individual sheet data
11
+ * - `xl/sharedStrings.xml` - Shared string table (cell text)
12
+ * - `xl/styles.xml` - Cell styling information
13
+ * - `xl/drawings/*` - Charts and drawings
14
+ * - `xl/media/*` - Embedded images
15
+ *
16
+ * **Key Elements:**
17
+ * - `<row>` - Table row with row index
18
+ * - `<c r="A1">` - Cell with reference (A1, B2, etc.)
19
+ * - `<v>` - Cell value (number or shared string index)
20
+ * - `<t="s">` - Cell type (s=string, n=number, b=boolean)
21
+ *
22
+ * @module ExcelParser
23
+ * @see https://www.ecma-international.org/publications-and-standards/standards/ecma-376/
24
+ */
25
+ Object.defineProperty(exports, "__esModule", { value: true });
26
+ exports.parseExcel = void 0;
27
+ const chartUtils_1 = require("../utils/chartUtils");
28
+ const errorUtils_1 = require("../utils/errorUtils");
29
+ const imageUtils_1 = require("../utils/imageUtils");
30
+ const ocrUtils_1 = require("../utils/ocrUtils");
31
+ const xmlUtils_1 = require("../utils/xmlUtils");
32
+ const zipUtils_1 = require("../utils/zipUtils");
33
+ /**
34
+ * Parses an Excel spreadsheet (.xlsx) and extracts sheets, rows, and cells.
35
+ *
36
+ * @param buffer - The XLSX file as a Buffer
37
+ * @param config - Parser configuration
38
+ * @returns A promise resolving to the parsed AST
39
+ */
40
+ const parseExcel = async (buffer, config) => {
41
+ const sheetsRegex = /xl\/worksheets\/sheet\d+.xml/g;
42
+ const drawingsRegex = /xl\/drawings\/drawing\d+.xml/g;
43
+ const chartsRegex = /xl\/charts\/chart\d+.xml/g;
44
+ const stringsFilePath = 'xl/sharedStrings.xml';
45
+ const mediaFileRegex = /xl\/media\/.*/;
46
+ const corePropsFileRegex = /docProps\/core\.xml/;
47
+ const relsRegex = /xl\/worksheets\/_rels\/sheet\d+\.xml\.rels/g;
48
+ const drawingRelsRegex = /xl\/drawings\/_rels\/drawing\d+\.xml\.rels/g;
49
+ const files = await (0, zipUtils_1.extractFiles)(buffer, (x) => !!x.match(sheetsRegex) ||
50
+ !!x.match(drawingsRegex) ||
51
+ !!x.match(chartsRegex) ||
52
+ x === stringsFilePath ||
53
+ x === 'xl/styles.xml' ||
54
+ x === 'xl/workbook.xml' ||
55
+ x === 'xl/_rels/workbook.xml.rels' ||
56
+ !!x.match(corePropsFileRegex) ||
57
+ (!!config.extractAttachments && (!!x.match(mediaFileRegex) || !!x.match(relsRegex) || !!x.match(drawingRelsRegex))));
58
+ const sharedStringsFile = files.find(f => f.path === stringsFilePath);
59
+ // Updated to store structured content (rich text runs) or simple string
60
+ const sharedStrings = [];
61
+ if (sharedStringsFile) {
62
+ const xml = (0, xmlUtils_1.parseXmlString)(sharedStringsFile.content.toString());
63
+ const siNodes = (0, xmlUtils_1.getElementsByTagName)(xml, "si");
64
+ for (const si of siNodes) {
65
+ const runNodes = (0, xmlUtils_1.getElementsByTagName)(si, "r");
66
+ if (runNodes.length > 0) {
67
+ // Rich text with runs
68
+ const runs = [];
69
+ for (const run of runNodes) {
70
+ const tNode = (0, xmlUtils_1.getElementsByTagName)(run, "t")[0];
71
+ if (tNode) {
72
+ const text = tNode.textContent || '';
73
+ // Extract run formatting
74
+ const rPr = (0, xmlUtils_1.getElementsByTagName)(run, "rPr")[0];
75
+ const formatting = {};
76
+ if (rPr) {
77
+ if ((0, xmlUtils_1.getElementsByTagName)(rPr, "b").length > 0)
78
+ formatting.bold = true;
79
+ if ((0, xmlUtils_1.getElementsByTagName)(rPr, "i").length > 0)
80
+ formatting.italic = true;
81
+ if ((0, xmlUtils_1.getElementsByTagName)(rPr, "u").length > 0)
82
+ formatting.underline = true;
83
+ if ((0, xmlUtils_1.getElementsByTagName)(rPr, "strike").length > 0)
84
+ formatting.strikethrough = true;
85
+ const sz = (0, xmlUtils_1.getElementsByTagName)(rPr, "sz")[0];
86
+ if (sz)
87
+ formatting.size = sz.getAttribute("val") + 'pt';
88
+ const color = (0, xmlUtils_1.getElementsByTagName)(rPr, "color")[0];
89
+ if (color) {
90
+ const rgb = color.getAttribute("rgb");
91
+ if (rgb)
92
+ formatting.color = '#' + rgb.substring(2);
93
+ }
94
+ const rFont = (0, xmlUtils_1.getElementsByTagName)(rPr, "rFont")[0];
95
+ if (rFont)
96
+ formatting.font = rFont.getAttribute("val") || undefined;
97
+ const vertAlign = (0, xmlUtils_1.getElementsByTagName)(rPr, "vertAlign")[0];
98
+ if (vertAlign) {
99
+ const val = vertAlign.getAttribute("val");
100
+ if (val === "subscript")
101
+ formatting.subscript = true;
102
+ if (val === "superscript")
103
+ formatting.superscript = true;
104
+ }
105
+ }
106
+ runs.push({
107
+ type: 'text',
108
+ text: text,
109
+ formatting: Object.keys(formatting).length > 0 ? formatting : undefined
110
+ });
111
+ }
112
+ }
113
+ sharedStrings.push(runs);
114
+ }
115
+ else {
116
+ // Simple text case
117
+ const tNodes = (0, xmlUtils_1.getElementsByTagName)(si, "t");
118
+ let text = '';
119
+ for (const t of tNodes) {
120
+ text += t.textContent || '';
121
+ }
122
+ sharedStrings.push(text);
123
+ }
124
+ }
125
+ }
126
+ // Parse styles to build formatting map
127
+ const stylesFile = files.find(f => f.path === 'xl/styles.xml');
128
+ const cellFormatMap = {};
129
+ if (stylesFile) {
130
+ const xml = (0, xmlUtils_1.parseXmlString)(stylesFile.content.toString());
131
+ // Parse fonts
132
+ const fontsNode = (0, xmlUtils_1.getElementsByTagName)(xml, "fonts")[0];
133
+ const fonts = [];
134
+ if (fontsNode) {
135
+ const fontNodes = (0, xmlUtils_1.getElementsByTagName)(fontsNode, "font");
136
+ for (const font of fontNodes) {
137
+ const formatting = {};
138
+ if ((0, xmlUtils_1.getElementsByTagName)(font, "b").length > 0)
139
+ formatting.bold = true;
140
+ if ((0, xmlUtils_1.getElementsByTagName)(font, "i").length > 0)
141
+ formatting.italic = true;
142
+ if ((0, xmlUtils_1.getElementsByTagName)(font, "u").length > 0)
143
+ formatting.underline = true;
144
+ if ((0, xmlUtils_1.getElementsByTagName)(font, "strike").length > 0)
145
+ formatting.strikethrough = true;
146
+ const szNode = (0, xmlUtils_1.getElementsByTagName)(font, "sz")[0];
147
+ if (szNode) {
148
+ const val = szNode.getAttribute("val");
149
+ if (val)
150
+ formatting.size = val + 'pt';
151
+ }
152
+ const colorNode = (0, xmlUtils_1.getElementsByTagName)(font, "color")[0];
153
+ if (colorNode) {
154
+ const rgb = colorNode.getAttribute("rgb");
155
+ if (rgb)
156
+ formatting.color = '#' + rgb.substring(2); // Remove alpha channel
157
+ }
158
+ const nameNode = (0, xmlUtils_1.getElementsByTagName)(font, "name")[0];
159
+ if (nameNode) {
160
+ const val = nameNode.getAttribute("val");
161
+ if (val)
162
+ formatting.font = val;
163
+ }
164
+ const vertAlignNode = (0, xmlUtils_1.getElementsByTagName)(font, "vertAlign")[0];
165
+ if (vertAlignNode) {
166
+ const val = vertAlignNode.getAttribute("val");
167
+ if (val === "subscript")
168
+ formatting.subscript = true;
169
+ if (val === "superscript")
170
+ formatting.superscript = true;
171
+ }
172
+ fonts.push(formatting);
173
+ }
174
+ }
175
+ // Parse fills (for background color)
176
+ const fillsNode = (0, xmlUtils_1.getElementsByTagName)(xml, "fills")[0];
177
+ const fills = [];
178
+ if (fillsNode) {
179
+ const fillNodes = (0, xmlUtils_1.getElementsByTagName)(fillsNode, "fill");
180
+ for (const fill of fillNodes) {
181
+ const formatting = {};
182
+ const patternFill = (0, xmlUtils_1.getElementsByTagName)(fill, "patternFill")[0];
183
+ if (patternFill) {
184
+ const fgColor = (0, xmlUtils_1.getElementsByTagName)(patternFill, "fgColor")[0];
185
+ if (fgColor) {
186
+ const rgb = fgColor.getAttribute("rgb");
187
+ const theme = fgColor.getAttribute("theme");
188
+ if (rgb && rgb !== "00000000") { // Not default/auto
189
+ formatting.backgroundColor = '#' + rgb.substring(2);
190
+ }
191
+ else if (theme) {
192
+ // Basic mapping for standard Office themes (Dark 1, Light 1, Dark 2, Light 2)
193
+ // 0: Light 1 (White), 1: Dark 1 (Black), 2: Light 2 (Tan/Gray), 3: Dark 2 (Blue/Grey)
194
+ const themeIdx = parseInt(theme);
195
+ if (themeIdx === 0)
196
+ formatting.backgroundColor = '#FFFFFF';
197
+ else if (themeIdx === 1)
198
+ formatting.backgroundColor = '#000000';
199
+ else if (themeIdx === 2)
200
+ formatting.backgroundColor = '#EEECE1'; // Standard Light 2
201
+ else if (themeIdx === 3)
202
+ formatting.backgroundColor = '#1F497D'; // Standard Dark 2
203
+ }
204
+ }
205
+ }
206
+ fills.push(formatting);
207
+ }
208
+ }
209
+ // Parse cellXfs (cell format definitions)
210
+ const cellXfsNode = (0, xmlUtils_1.getElementsByTagName)(xml, "cellXfs")[0];
211
+ if (cellXfsNode) {
212
+ const xfNodes = (0, xmlUtils_1.getElementsByTagName)(cellXfsNode, "xf");
213
+ for (let i = 0; i < xfNodes.length; i++) {
214
+ const xf = xfNodes[i];
215
+ const formatting = {};
216
+ const fontId = xf.getAttribute("fontId");
217
+ if (fontId) {
218
+ const fontIdx = parseInt(fontId);
219
+ if (fonts[fontIdx]) {
220
+ Object.assign(formatting, fonts[fontIdx]);
221
+ }
222
+ }
223
+ const fillId = xf.getAttribute("fillId");
224
+ if (fillId) {
225
+ const fillIdx = parseInt(fillId);
226
+ if (fills[fillIdx] && fills[fillIdx].backgroundColor) {
227
+ formatting.backgroundColor = fills[fillIdx].backgroundColor;
228
+ }
229
+ }
230
+ const alignmentNode = (0, xmlUtils_1.getElementsByTagName)(xf, "alignment")[0];
231
+ if (alignmentNode) {
232
+ const horizontal = alignmentNode.getAttribute("horizontal");
233
+ if (horizontal === 'center' || horizontal === 'right' || horizontal === 'justify' || horizontal === 'left') {
234
+ formatting.alignment = horizontal;
235
+ }
236
+ }
237
+ cellFormatMap[i] = formatting;
238
+ }
239
+ }
240
+ }
241
+ const attachments = [];
242
+ const mediaFiles = files.filter(f => f.path.match(/xl\/media\/.*/));
243
+ const chartFiles = files.filter(f => f.path.match(chartsRegex));
244
+ // Map to store image details by drawing file path and relationship ID
245
+ const drawingImageMap = {};
246
+ if (config.extractAttachments) {
247
+ // 1. Parse Drawing Rels to map rIds to media paths
248
+ const drawingRelsFiles = files.filter(f => f.path.match(drawingRelsRegex));
249
+ for (const relFile of drawingRelsFiles) {
250
+ const drawingFilename = relFile.path.split('/').pop()?.replace('.rels', '') || '';
251
+ const drawingPath = `xl/drawings/${drawingFilename}`;
252
+ const relsXml = (0, xmlUtils_1.parseXmlString)(relFile.content.toString());
253
+ const relationships = (0, xmlUtils_1.getElementsByTagName)(relsXml, "Relationship");
254
+ if (!drawingImageMap[drawingPath]) {
255
+ drawingImageMap[drawingPath] = {};
256
+ }
257
+ for (const rel of relationships) {
258
+ const id = rel.getAttribute("Id");
259
+ const target = rel.getAttribute("Target");
260
+ if (id && target && target.includes('media/')) {
261
+ // Target is usually like "../media/image1.png"
262
+ const mediaPath = 'xl/' + target.replace('../', '');
263
+ drawingImageMap[drawingPath][id] = { path: mediaPath };
264
+ }
265
+ }
266
+ }
267
+ // 2. Parse Drawings to get Alt Text and link to Rels
268
+ const drawingFiles = files.filter(f => f.path.match(drawingsRegex));
269
+ for (const drawingFile of drawingFiles) {
270
+ const xml = (0, xmlUtils_1.parseXmlString)(drawingFile.content.toString());
271
+ const pics = (0, xmlUtils_1.getElementsByTagName)(xml, "xdr:pic"); // SpreadsheetML drawing
272
+ const rels = drawingImageMap[drawingFile.path] || {};
273
+ for (const pic of pics) {
274
+ const blipFill = (0, xmlUtils_1.getElementsByTagName)(pic, "xdr:blipFill")[0];
275
+ const blip = blipFill ? (0, xmlUtils_1.getElementsByTagName)(blipFill, "a:blip")[0] : null;
276
+ const embedId = blip ? blip.getAttribute("r:embed") : null;
277
+ const nvPicPr = (0, xmlUtils_1.getElementsByTagName)(pic, "xdr:nvPicPr")[0];
278
+ const cNvPr = nvPicPr ? (0, xmlUtils_1.getElementsByTagName)(nvPicPr, "xdr:cNvPr")[0] : null;
279
+ const altText = cNvPr ? (cNvPr.getAttribute("descr") || cNvPr.getAttribute("name")) : undefined;
280
+ if (embedId && rels[embedId]) {
281
+ rels[embedId].altText = altText || '';
282
+ }
283
+ }
284
+ }
285
+ // 3. Process Media Files
286
+ for (const media of mediaFiles) {
287
+ const attachment = (0, imageUtils_1.createAttachment)(media.path.split('/').pop() || 'image', media.content);
288
+ // Try to find alt text for this media
289
+ let altText = '';
290
+ for (const drawingPath in drawingImageMap) {
291
+ for (const rId in drawingImageMap[drawingPath]) {
292
+ if (drawingImageMap[drawingPath][rId].path === media.path) {
293
+ altText = drawingImageMap[drawingPath][rId].altText || '';
294
+ break;
295
+ }
296
+ }
297
+ if (altText)
298
+ break;
299
+ }
300
+ if (altText)
301
+ attachment.altText = altText;
302
+ attachments.push(attachment);
303
+ if (config.ocr) {
304
+ if (attachment.mimeType.startsWith('image/')) {
305
+ try {
306
+ const ocrText = await (0, ocrUtils_1.performOcr)(media.content, config.ocrLanguage);
307
+ if (ocrText.trim()) {
308
+ attachment.ocrText = ocrText.trim();
309
+ }
310
+ }
311
+ catch (e) {
312
+ (0, errorUtils_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
313
+ }
314
+ }
315
+ }
316
+ }
317
+ for (const chart of chartFiles) {
318
+ const attachment = {
319
+ type: 'chart',
320
+ mimeType: 'application/vnd.openxmlformats-officedocument.wordprocessingml.document',
321
+ data: chart.content.toString('base64'),
322
+ name: chart.path.split('/').pop() || '',
323
+ extension: 'xml'
324
+ };
325
+ // Extract structured chart data
326
+ try {
327
+ const chartData = (0, chartUtils_1.extractChartData)(chart.content);
328
+ attachment.chartData = chartData;
329
+ }
330
+ catch (e) {
331
+ (0, errorUtils_1.logWarning)(`Failed to extract chart data from ${chart.path}:`, config, e);
332
+ }
333
+ attachments.push(attachment);
334
+ }
335
+ }
336
+ // Build map of drawing rId -> chart attachment name for linking
337
+ const drawingChartMap = {};
338
+ if (config.extractAttachments) {
339
+ const drawingRelsFiles = files.filter(f => f.path.match(drawingRelsRegex));
340
+ for (const relFile of drawingRelsFiles) {
341
+ const drawingFilename = relFile.path.split('/').pop()?.replace('.rels', '') || '';
342
+ const drawingPath = `xl/drawings/${drawingFilename}`;
343
+ const relsXml = (0, xmlUtils_1.parseXmlString)(relFile.content.toString());
344
+ const relationships = (0, xmlUtils_1.getElementsByTagName)(relsXml, "Relationship");
345
+ if (!drawingChartMap[drawingPath]) {
346
+ drawingChartMap[drawingPath] = {};
347
+ }
348
+ for (const rel of relationships) {
349
+ const id = rel.getAttribute("Id");
350
+ const target = rel.getAttribute("Target");
351
+ const type = rel.getAttribute("Type");
352
+ if (id && target && type && type.includes('chart')) {
353
+ // Target is like "../charts/chart1.xml"
354
+ const chartName = target.split('/').pop() || '';
355
+ drawingChartMap[drawingPath][id] = chartName;
356
+ }
357
+ }
358
+ }
359
+ }
360
+ // Parse workbook.xml to get sheet names and map them to sheet files
361
+ const sheetNameMap = {};
362
+ const workbookFile = files.find(f => f.path === 'xl/workbook.xml');
363
+ const workbookRelsFile = files.find(f => f.path === 'xl/_rels/workbook.xml.rels');
364
+ if (workbookFile && workbookRelsFile) {
365
+ // Parse rels to get rId -> file mapping
366
+ const relsXml = (0, xmlUtils_1.parseXmlString)(workbookRelsFile.content.toString());
367
+ const relationships = (0, xmlUtils_1.getElementsByTagName)(relsXml, "Relationship");
368
+ const rIdToFile = {};
369
+ for (const rel of relationships) {
370
+ const rId = rel.getAttribute("Id");
371
+ const target = rel.getAttribute("Target");
372
+ if (rId && target) {
373
+ // Target is like "worksheets/sheet1.xml"
374
+ const filename = target.split('/').pop() || '';
375
+ rIdToFile[rId] = filename;
376
+ }
377
+ }
378
+ // Parse workbook.xml to get sheet name -> rId mapping
379
+ const workbookXml = (0, xmlUtils_1.parseXmlString)(workbookFile.content.toString());
380
+ const sheets = (0, xmlUtils_1.getElementsByTagName)(workbookXml, "sheet");
381
+ for (const sheet of sheets) {
382
+ const name = sheet.getAttribute("name");
383
+ const rId = sheet.getAttribute("r:id");
384
+ if (name && rId && rIdToFile[rId]) {
385
+ sheetNameMap[rIdToFile[rId]] = name;
386
+ }
387
+ }
388
+ }
389
+ const content = [];
390
+ const rawContents = [];
391
+ for (const file of files) {
392
+ if (file.path.match(mediaFileRegex))
393
+ continue;
394
+ if (file.path === stringsFilePath)
395
+ continue;
396
+ if (file.path === 'xl/styles.xml')
397
+ continue;
398
+ if (file.path.match(drawingsRegex))
399
+ continue;
400
+ if (file.path.match(chartsRegex))
401
+ continue;
402
+ if (file.path.match(relsRegex))
403
+ continue;
404
+ if (file.path.match(drawingRelsRegex))
405
+ continue;
406
+ if (file.path.match(sheetsRegex)) {
407
+ if (config.includeRawContent) {
408
+ rawContents.push(file.content.toString());
409
+ }
410
+ const rows = [];
411
+ const rowRegex = /<row.*?>[\s\S]*?<\/row>/g;
412
+ const rowMatches = file.content.toString().match(rowRegex);
413
+ if (rowMatches) {
414
+ for (const rowXml of rowMatches) {
415
+ const cells = [];
416
+ const cRegex = /<c.*?>[\s\S]*?<\/c>/g;
417
+ const cMatches = rowXml.match(cRegex);
418
+ const rMatch = rowXml.match(/r="(\d+)"/);
419
+ const rowIndex = rMatch ? parseInt(rMatch[1]) - 1 : 0;
420
+ if (cMatches) {
421
+ for (const cXml of cMatches) {
422
+ // Extract cell value
423
+ const typeMatch = cXml.match(/t="([a-z]+)"/);
424
+ const type = typeMatch ? typeMatch[1] : 'n'; // n = number (default)
425
+ const vMatch = cXml.match(/<v>(.*?)<\/v>/);
426
+ const tMatch = cXml.match(/<t>(.*?)<\/t>/);
427
+ let text = '';
428
+ let cellNodes = [];
429
+ if (type === 's' && vMatch) {
430
+ const idx = parseInt(vMatch[1]);
431
+ const content = sharedStrings[idx];
432
+ if (Array.isArray(content)) {
433
+ // Rich text runs
434
+ // Deep copy runs to avoid reference issues if reused
435
+ cellNodes = JSON.parse(JSON.stringify(content));
436
+ text = cellNodes.map(n => n.text).join('');
437
+ }
438
+ else {
439
+ text = content || '';
440
+ }
441
+ }
442
+ else if (type === 'inlineStr' && tMatch) {
443
+ text = tMatch[1];
444
+ }
445
+ else if (vMatch) {
446
+ text = vMatch[1];
447
+ }
448
+ // Parse cell coordinate
449
+ const coordMatch = cXml.match(/r="([A-Z]+)(\d+)"/);
450
+ const colStr = coordMatch ? coordMatch[1] : '';
451
+ const colIndex = colStr.charCodeAt(0) - 'A'.charCodeAt(0);
452
+ if (text || cellNodes.length > 0) {
453
+ // Extract cell style index
454
+ const styleMatch = cXml.match(/s="(\d+)"/);
455
+ const styleIdx = styleMatch ? parseInt(styleMatch[1]) : undefined;
456
+ const cellFormatting = (styleIdx !== undefined && cellFormatMap[styleIdx]) ? cellFormatMap[styleIdx] : {};
457
+ if (cellNodes.length > 0) {
458
+ // If we have specific runs, merge cell styles into them if run style is missing
459
+ // But usually run style overrides cell style (except maybe background)
460
+ for (const node of cellNodes) {
461
+ if (!node.formatting)
462
+ node.formatting = {};
463
+ // Cell background always applies
464
+ if (cellFormatting.backgroundColor)
465
+ node.formatting.backgroundColor = cellFormatting.backgroundColor;
466
+ // Cell alignment always applies
467
+ if (cellFormatting.alignment)
468
+ node.formatting.alignment = cellFormatting.alignment;
469
+ // Font defaults from cell style if not in run
470
+ if (!node.formatting.font && cellFormatting.font)
471
+ node.formatting.font = cellFormatting.font;
472
+ if (!node.formatting.size && cellFormatting.size)
473
+ node.formatting.size = cellFormatting.size;
474
+ }
475
+ }
476
+ else {
477
+ // Simple text node
478
+ cellNodes.push({
479
+ type: 'text',
480
+ text: text,
481
+ formatting: cellFormatting
482
+ });
483
+ }
484
+ const cellNode = {
485
+ type: 'cell',
486
+ text: text,
487
+ children: cellNodes,
488
+ metadata: { row: rowIndex, col: colIndex }
489
+ };
490
+ if (config.includeRawContent) {
491
+ cellNode.rawContent = cXml;
492
+ }
493
+ cells.push(cellNode);
494
+ }
495
+ }
496
+ }
497
+ if (cells.length > 0) {
498
+ const rowNode = {
499
+ type: 'row',
500
+ children: cells,
501
+ metadata: undefined
502
+ };
503
+ if (config.includeRawContent) {
504
+ rowNode.rawContent = rowXml;
505
+ }
506
+ rows.push(rowNode);
507
+ }
508
+ }
509
+ }
510
+ // Handle Drawings in Sheet (images and charts)
511
+ if (config.extractAttachments) {
512
+ // Parse Sheet Rels to map drawing rIds
513
+ const sheetFilename = file.path.split('/').pop() || '';
514
+ const relsFilename = `xl/worksheets/_rels/${sheetFilename}.rels`;
515
+ const relsFile = files.find(f => f.path === relsFilename);
516
+ const drawingMap = {}; // rId -> drawingPath
517
+ if (relsFile) {
518
+ const relsXml = (0, xmlUtils_1.parseXmlString)(relsFile.content.toString());
519
+ const relationships = (0, xmlUtils_1.getElementsByTagName)(relsXml, "Relationship");
520
+ for (const rel of relationships) {
521
+ const id = rel.getAttribute("Id");
522
+ const target = rel.getAttribute("Target");
523
+ const type = rel.getAttribute("Type");
524
+ if (id && target && type && type.includes('drawing')) {
525
+ drawingMap[id] = 'xl/drawings/' + target.replace('../drawings/', '');
526
+ }
527
+ }
528
+ }
529
+ const drawingMatches = file.content.toString().match(/<drawing r:id="(.*?)"/g);
530
+ if (drawingMatches) {
531
+ for (const match of drawingMatches) {
532
+ const rIdMatch = match.match(/r:id="(.*?)"/);
533
+ const rId = rIdMatch ? rIdMatch[1] : null;
534
+ if (rId && drawingMap[rId]) {
535
+ const drawingPath = drawingMap[rId];
536
+ // Find all images in this drawing
537
+ const images = drawingImageMap[drawingPath];
538
+ if (images) {
539
+ for (const imgId in images) {
540
+ const imgInfo = images[imgId];
541
+ const attachment = attachments.find(a => a.name === imgInfo.path.split('/').pop());
542
+ if (attachment) {
543
+ const imageNode = {
544
+ type: 'image',
545
+ text: '',
546
+ children: [],
547
+ metadata: {
548
+ attachmentName: attachment.name || 'unknown',
549
+ altText: imgInfo.altText || undefined
550
+ }
551
+ };
552
+ rows.push(imageNode);
553
+ }
554
+ }
555
+ }
556
+ // Find all charts in this drawing
557
+ const charts = drawingChartMap[drawingPath];
558
+ if (charts) {
559
+ for (const chartRId in charts) {
560
+ const chartName = charts[chartRId];
561
+ const attachment = attachments.find(a => a.name === chartName);
562
+ if (attachment) {
563
+ const chartNode = {
564
+ type: 'chart',
565
+ text: '',
566
+ children: [],
567
+ metadata: {
568
+ attachmentName: chartName
569
+ }
570
+ };
571
+ rows.push(chartNode);
572
+ }
573
+ }
574
+ }
575
+ }
576
+ }
577
+ }
578
+ }
579
+ // Get proper sheet name from workbook.xml mapping, fallback to filename
580
+ const sheetFileName = file.path.split('/').pop() || 'Sheet';
581
+ const sheetName = sheetNameMap[sheetFileName] || sheetFileName;
582
+ content.push({
583
+ type: 'sheet',
584
+ children: rows,
585
+ metadata: { sheetName },
586
+ rawContent: config.includeRawContent ? file.content.toString() : undefined
587
+ });
588
+ }
589
+ }
590
+ const corePropsFile = files.find(f => f.path.match(corePropsFileRegex));
591
+ const metadata = corePropsFile ? (0, xmlUtils_1.parseOfficeMetadata)(corePropsFile.content.toString()) : {};
592
+ // Link OCR text and chart data to content nodes (like PPTX parser)
593
+ const assignAttachmentData = (nodes) => {
594
+ for (const node of nodes) {
595
+ if ('attachmentName' in (node.metadata || {})) {
596
+ const meta = node.metadata;
597
+ const attachment = attachments.find(a => a.name === meta.attachmentName);
598
+ if (attachment) {
599
+ if (node.type === 'image') {
600
+ // Link OCR text to image node
601
+ if (attachment.ocrText) {
602
+ node.text = attachment.ocrText;
603
+ }
604
+ // Copy altText to attachment
605
+ if (meta.altText) {
606
+ attachment.altText = meta.altText;
607
+ }
608
+ }
609
+ if (node.type === 'chart') {
610
+ // Link chart data text to chart node
611
+ if (attachment.chartData) {
612
+ node.text = attachment.chartData.rawTexts.join(config.newlineDelimiter || '\n');
613
+ }
614
+ }
615
+ }
616
+ }
617
+ if (node.children) {
618
+ assignAttachmentData(node.children);
619
+ }
620
+ }
621
+ };
622
+ assignAttachmentData(content);
623
+ return {
624
+ type: 'xlsx',
625
+ metadata: metadata,
626
+ content: content,
627
+ attachments: attachments,
628
+ toText: () => content.map(c => {
629
+ // Recursive text extraction
630
+ const getText = (node) => {
631
+ let t = '';
632
+ if (node.children) {
633
+ t += node.children.map(getText).filter(t => t != '').join(!node.children[0]?.children ? '' : config.newlineDelimiter ?? '\n');
634
+ }
635
+ else
636
+ t += node.text || '';
637
+ return t;
638
+ };
639
+ return getText(c);
640
+ }).filter(t => t != '').join(config.newlineDelimiter ?? '\n')
641
+ };
642
+ };
643
+ exports.parseExcel = parseExcel;