officeparser 5.2.1 → 6.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +411 -163
- package/dist/OfficeParser.d.ts +90 -0
- package/dist/OfficeParser.js +217 -0
- package/dist/index.d.ts +51 -0
- package/dist/index.js +108 -0
- package/dist/officeparser.browser.js +165 -0
- package/dist/officeparser.browser.js.map +7 -0
- package/dist/parsers/ExcelParser.d.ts +33 -0
- package/dist/parsers/ExcelParser.js +643 -0
- package/dist/parsers/OpenOfficeParser.d.ts +32 -0
- package/dist/parsers/OpenOfficeParser.js +1399 -0
- package/dist/parsers/PdfParser.d.ts +68 -0
- package/dist/parsers/PdfParser.js +847 -0
- package/dist/parsers/PowerPointParser.d.ts +33 -0
- package/dist/parsers/PowerPointParser.js +778 -0
- package/dist/parsers/RtfParser.d.ts +164 -0
- package/dist/parsers/RtfParser.js +1641 -0
- package/dist/parsers/WordParser.d.ts +79 -0
- package/dist/parsers/WordParser.js +787 -0
- package/dist/types.d.ts +615 -0
- package/dist/types.js +2 -0
- package/dist/utils/chartUtils.d.ts +7 -0
- package/dist/utils/chartUtils.js +255 -0
- package/dist/utils/errorUtils.d.ts +58 -0
- package/dist/utils/errorUtils.js +120 -0
- package/dist/utils/imageUtils.d.ts +67 -0
- package/dist/utils/imageUtils.js +133 -0
- package/dist/utils/ocrUtils.d.ts +39 -0
- package/dist/utils/ocrUtils.js +61 -0
- package/dist/utils/xmlUtils.d.ts +83 -0
- package/dist/utils/xmlUtils.js +158 -0
- package/dist/utils/zipUtils.d.ts +74 -0
- package/dist/utils/zipUtils.js +112 -0
- package/package.json +44 -17
- package/officeParser.js +0 -776
- package/typings/officeParser.d.ts +0 -33
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Excel Spreadsheet (XLSX) Parser
|
|
3
|
+
*
|
|
4
|
+
* **XLSX Format Overview:**
|
|
5
|
+
* XLSX is the default format for Microsoft Excel since Office 2007, based on OOXML.
|
|
6
|
+
*
|
|
7
|
+
* **File Structure:**
|
|
8
|
+
* - `xl/workbook.xml` - Workbook structure and sheet list
|
|
9
|
+
* - `xl/worksheets/sheet1.xml` - Individual sheet data
|
|
10
|
+
* - `xl/sharedStrings.xml` - Shared string table (cell text)
|
|
11
|
+
* - `xl/styles.xml` - Cell styling information
|
|
12
|
+
* - `xl/drawings/*` - Charts and drawings
|
|
13
|
+
* - `xl/media/*` - Embedded images
|
|
14
|
+
*
|
|
15
|
+
* **Key Elements:**
|
|
16
|
+
* - `<row>` - Table row with row index
|
|
17
|
+
* - `<c r="A1">` - Cell with reference (A1, B2, etc.)
|
|
18
|
+
* - `<v>` - Cell value (number or shared string index)
|
|
19
|
+
* - `<t="s">` - Cell type (s=string, n=number, b=boolean)
|
|
20
|
+
*
|
|
21
|
+
* @module ExcelParser
|
|
22
|
+
* @see https://www.ecma-international.org/publications-and-standards/standards/ecma-376/
|
|
23
|
+
*/
|
|
24
|
+
/// <reference types="node" />
|
|
25
|
+
import { OfficeParserAST, OfficeParserConfig } from '../types';
|
|
26
|
+
/**
|
|
27
|
+
* Parses an Excel spreadsheet (.xlsx) and extracts sheets, rows, and cells.
|
|
28
|
+
*
|
|
29
|
+
* @param buffer - The XLSX file as a Buffer
|
|
30
|
+
* @param config - Parser configuration
|
|
31
|
+
* @returns A promise resolving to the parsed AST
|
|
32
|
+
*/
|
|
33
|
+
export declare const parseExcel: (buffer: Buffer, config: OfficeParserConfig) => Promise<OfficeParserAST>;
|
|
@@ -0,0 +1,643 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* Excel Spreadsheet (XLSX) Parser
|
|
4
|
+
*
|
|
5
|
+
* **XLSX Format Overview:**
|
|
6
|
+
* XLSX is the default format for Microsoft Excel since Office 2007, based on OOXML.
|
|
7
|
+
*
|
|
8
|
+
* **File Structure:**
|
|
9
|
+
* - `xl/workbook.xml` - Workbook structure and sheet list
|
|
10
|
+
* - `xl/worksheets/sheet1.xml` - Individual sheet data
|
|
11
|
+
* - `xl/sharedStrings.xml` - Shared string table (cell text)
|
|
12
|
+
* - `xl/styles.xml` - Cell styling information
|
|
13
|
+
* - `xl/drawings/*` - Charts and drawings
|
|
14
|
+
* - `xl/media/*` - Embedded images
|
|
15
|
+
*
|
|
16
|
+
* **Key Elements:**
|
|
17
|
+
* - `<row>` - Table row with row index
|
|
18
|
+
* - `<c r="A1">` - Cell with reference (A1, B2, etc.)
|
|
19
|
+
* - `<v>` - Cell value (number or shared string index)
|
|
20
|
+
* - `<t="s">` - Cell type (s=string, n=number, b=boolean)
|
|
21
|
+
*
|
|
22
|
+
* @module ExcelParser
|
|
23
|
+
* @see https://www.ecma-international.org/publications-and-standards/standards/ecma-376/
|
|
24
|
+
*/
|
|
25
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
26
|
+
exports.parseExcel = void 0;
|
|
27
|
+
const chartUtils_1 = require("../utils/chartUtils");
|
|
28
|
+
const errorUtils_1 = require("../utils/errorUtils");
|
|
29
|
+
const imageUtils_1 = require("../utils/imageUtils");
|
|
30
|
+
const ocrUtils_1 = require("../utils/ocrUtils");
|
|
31
|
+
const xmlUtils_1 = require("../utils/xmlUtils");
|
|
32
|
+
const zipUtils_1 = require("../utils/zipUtils");
|
|
33
|
+
/**
|
|
34
|
+
* Parses an Excel spreadsheet (.xlsx) and extracts sheets, rows, and cells.
|
|
35
|
+
*
|
|
36
|
+
* @param buffer - The XLSX file as a Buffer
|
|
37
|
+
* @param config - Parser configuration
|
|
38
|
+
* @returns A promise resolving to the parsed AST
|
|
39
|
+
*/
|
|
40
|
+
const parseExcel = async (buffer, config) => {
|
|
41
|
+
const sheetsRegex = /xl\/worksheets\/sheet\d+.xml/g;
|
|
42
|
+
const drawingsRegex = /xl\/drawings\/drawing\d+.xml/g;
|
|
43
|
+
const chartsRegex = /xl\/charts\/chart\d+.xml/g;
|
|
44
|
+
const stringsFilePath = 'xl/sharedStrings.xml';
|
|
45
|
+
const mediaFileRegex = /xl\/media\/.*/;
|
|
46
|
+
const corePropsFileRegex = /docProps\/core\.xml/;
|
|
47
|
+
const relsRegex = /xl\/worksheets\/_rels\/sheet\d+\.xml\.rels/g;
|
|
48
|
+
const drawingRelsRegex = /xl\/drawings\/_rels\/drawing\d+\.xml\.rels/g;
|
|
49
|
+
const files = await (0, zipUtils_1.extractFiles)(buffer, (x) => !!x.match(sheetsRegex) ||
|
|
50
|
+
!!x.match(drawingsRegex) ||
|
|
51
|
+
!!x.match(chartsRegex) ||
|
|
52
|
+
x === stringsFilePath ||
|
|
53
|
+
x === 'xl/styles.xml' ||
|
|
54
|
+
x === 'xl/workbook.xml' ||
|
|
55
|
+
x === 'xl/_rels/workbook.xml.rels' ||
|
|
56
|
+
!!x.match(corePropsFileRegex) ||
|
|
57
|
+
(!!config.extractAttachments && (!!x.match(mediaFileRegex) || !!x.match(relsRegex) || !!x.match(drawingRelsRegex))));
|
|
58
|
+
const sharedStringsFile = files.find(f => f.path === stringsFilePath);
|
|
59
|
+
// Updated to store structured content (rich text runs) or simple string
|
|
60
|
+
const sharedStrings = [];
|
|
61
|
+
if (sharedStringsFile) {
|
|
62
|
+
const xml = (0, xmlUtils_1.parseXmlString)(sharedStringsFile.content.toString());
|
|
63
|
+
const siNodes = (0, xmlUtils_1.getElementsByTagName)(xml, "si");
|
|
64
|
+
for (const si of siNodes) {
|
|
65
|
+
const runNodes = (0, xmlUtils_1.getElementsByTagName)(si, "r");
|
|
66
|
+
if (runNodes.length > 0) {
|
|
67
|
+
// Rich text with runs
|
|
68
|
+
const runs = [];
|
|
69
|
+
for (const run of runNodes) {
|
|
70
|
+
const tNode = (0, xmlUtils_1.getElementsByTagName)(run, "t")[0];
|
|
71
|
+
if (tNode) {
|
|
72
|
+
const text = tNode.textContent || '';
|
|
73
|
+
// Extract run formatting
|
|
74
|
+
const rPr = (0, xmlUtils_1.getElementsByTagName)(run, "rPr")[0];
|
|
75
|
+
const formatting = {};
|
|
76
|
+
if (rPr) {
|
|
77
|
+
if ((0, xmlUtils_1.getElementsByTagName)(rPr, "b").length > 0)
|
|
78
|
+
formatting.bold = true;
|
|
79
|
+
if ((0, xmlUtils_1.getElementsByTagName)(rPr, "i").length > 0)
|
|
80
|
+
formatting.italic = true;
|
|
81
|
+
if ((0, xmlUtils_1.getElementsByTagName)(rPr, "u").length > 0)
|
|
82
|
+
formatting.underline = true;
|
|
83
|
+
if ((0, xmlUtils_1.getElementsByTagName)(rPr, "strike").length > 0)
|
|
84
|
+
formatting.strikethrough = true;
|
|
85
|
+
const sz = (0, xmlUtils_1.getElementsByTagName)(rPr, "sz")[0];
|
|
86
|
+
if (sz)
|
|
87
|
+
formatting.size = sz.getAttribute("val") + 'pt';
|
|
88
|
+
const color = (0, xmlUtils_1.getElementsByTagName)(rPr, "color")[0];
|
|
89
|
+
if (color) {
|
|
90
|
+
const rgb = color.getAttribute("rgb");
|
|
91
|
+
if (rgb)
|
|
92
|
+
formatting.color = '#' + rgb.substring(2);
|
|
93
|
+
}
|
|
94
|
+
const rFont = (0, xmlUtils_1.getElementsByTagName)(rPr, "rFont")[0];
|
|
95
|
+
if (rFont)
|
|
96
|
+
formatting.font = rFont.getAttribute("val") || undefined;
|
|
97
|
+
const vertAlign = (0, xmlUtils_1.getElementsByTagName)(rPr, "vertAlign")[0];
|
|
98
|
+
if (vertAlign) {
|
|
99
|
+
const val = vertAlign.getAttribute("val");
|
|
100
|
+
if (val === "subscript")
|
|
101
|
+
formatting.subscript = true;
|
|
102
|
+
if (val === "superscript")
|
|
103
|
+
formatting.superscript = true;
|
|
104
|
+
}
|
|
105
|
+
}
|
|
106
|
+
runs.push({
|
|
107
|
+
type: 'text',
|
|
108
|
+
text: text,
|
|
109
|
+
formatting: Object.keys(formatting).length > 0 ? formatting : undefined
|
|
110
|
+
});
|
|
111
|
+
}
|
|
112
|
+
}
|
|
113
|
+
sharedStrings.push(runs);
|
|
114
|
+
}
|
|
115
|
+
else {
|
|
116
|
+
// Simple text case
|
|
117
|
+
const tNodes = (0, xmlUtils_1.getElementsByTagName)(si, "t");
|
|
118
|
+
let text = '';
|
|
119
|
+
for (const t of tNodes) {
|
|
120
|
+
text += t.textContent || '';
|
|
121
|
+
}
|
|
122
|
+
sharedStrings.push(text);
|
|
123
|
+
}
|
|
124
|
+
}
|
|
125
|
+
}
|
|
126
|
+
// Parse styles to build formatting map
|
|
127
|
+
const stylesFile = files.find(f => f.path === 'xl/styles.xml');
|
|
128
|
+
const cellFormatMap = {};
|
|
129
|
+
if (stylesFile) {
|
|
130
|
+
const xml = (0, xmlUtils_1.parseXmlString)(stylesFile.content.toString());
|
|
131
|
+
// Parse fonts
|
|
132
|
+
const fontsNode = (0, xmlUtils_1.getElementsByTagName)(xml, "fonts")[0];
|
|
133
|
+
const fonts = [];
|
|
134
|
+
if (fontsNode) {
|
|
135
|
+
const fontNodes = (0, xmlUtils_1.getElementsByTagName)(fontsNode, "font");
|
|
136
|
+
for (const font of fontNodes) {
|
|
137
|
+
const formatting = {};
|
|
138
|
+
if ((0, xmlUtils_1.getElementsByTagName)(font, "b").length > 0)
|
|
139
|
+
formatting.bold = true;
|
|
140
|
+
if ((0, xmlUtils_1.getElementsByTagName)(font, "i").length > 0)
|
|
141
|
+
formatting.italic = true;
|
|
142
|
+
if ((0, xmlUtils_1.getElementsByTagName)(font, "u").length > 0)
|
|
143
|
+
formatting.underline = true;
|
|
144
|
+
if ((0, xmlUtils_1.getElementsByTagName)(font, "strike").length > 0)
|
|
145
|
+
formatting.strikethrough = true;
|
|
146
|
+
const szNode = (0, xmlUtils_1.getElementsByTagName)(font, "sz")[0];
|
|
147
|
+
if (szNode) {
|
|
148
|
+
const val = szNode.getAttribute("val");
|
|
149
|
+
if (val)
|
|
150
|
+
formatting.size = val + 'pt';
|
|
151
|
+
}
|
|
152
|
+
const colorNode = (0, xmlUtils_1.getElementsByTagName)(font, "color")[0];
|
|
153
|
+
if (colorNode) {
|
|
154
|
+
const rgb = colorNode.getAttribute("rgb");
|
|
155
|
+
if (rgb)
|
|
156
|
+
formatting.color = '#' + rgb.substring(2); // Remove alpha channel
|
|
157
|
+
}
|
|
158
|
+
const nameNode = (0, xmlUtils_1.getElementsByTagName)(font, "name")[0];
|
|
159
|
+
if (nameNode) {
|
|
160
|
+
const val = nameNode.getAttribute("val");
|
|
161
|
+
if (val)
|
|
162
|
+
formatting.font = val;
|
|
163
|
+
}
|
|
164
|
+
const vertAlignNode = (0, xmlUtils_1.getElementsByTagName)(font, "vertAlign")[0];
|
|
165
|
+
if (vertAlignNode) {
|
|
166
|
+
const val = vertAlignNode.getAttribute("val");
|
|
167
|
+
if (val === "subscript")
|
|
168
|
+
formatting.subscript = true;
|
|
169
|
+
if (val === "superscript")
|
|
170
|
+
formatting.superscript = true;
|
|
171
|
+
}
|
|
172
|
+
fonts.push(formatting);
|
|
173
|
+
}
|
|
174
|
+
}
|
|
175
|
+
// Parse fills (for background color)
|
|
176
|
+
const fillsNode = (0, xmlUtils_1.getElementsByTagName)(xml, "fills")[0];
|
|
177
|
+
const fills = [];
|
|
178
|
+
if (fillsNode) {
|
|
179
|
+
const fillNodes = (0, xmlUtils_1.getElementsByTagName)(fillsNode, "fill");
|
|
180
|
+
for (const fill of fillNodes) {
|
|
181
|
+
const formatting = {};
|
|
182
|
+
const patternFill = (0, xmlUtils_1.getElementsByTagName)(fill, "patternFill")[0];
|
|
183
|
+
if (patternFill) {
|
|
184
|
+
const fgColor = (0, xmlUtils_1.getElementsByTagName)(patternFill, "fgColor")[0];
|
|
185
|
+
if (fgColor) {
|
|
186
|
+
const rgb = fgColor.getAttribute("rgb");
|
|
187
|
+
const theme = fgColor.getAttribute("theme");
|
|
188
|
+
if (rgb && rgb !== "00000000") { // Not default/auto
|
|
189
|
+
formatting.backgroundColor = '#' + rgb.substring(2);
|
|
190
|
+
}
|
|
191
|
+
else if (theme) {
|
|
192
|
+
// Basic mapping for standard Office themes (Dark 1, Light 1, Dark 2, Light 2)
|
|
193
|
+
// 0: Light 1 (White), 1: Dark 1 (Black), 2: Light 2 (Tan/Gray), 3: Dark 2 (Blue/Grey)
|
|
194
|
+
const themeIdx = parseInt(theme);
|
|
195
|
+
if (themeIdx === 0)
|
|
196
|
+
formatting.backgroundColor = '#FFFFFF';
|
|
197
|
+
else if (themeIdx === 1)
|
|
198
|
+
formatting.backgroundColor = '#000000';
|
|
199
|
+
else if (themeIdx === 2)
|
|
200
|
+
formatting.backgroundColor = '#EEECE1'; // Standard Light 2
|
|
201
|
+
else if (themeIdx === 3)
|
|
202
|
+
formatting.backgroundColor = '#1F497D'; // Standard Dark 2
|
|
203
|
+
}
|
|
204
|
+
}
|
|
205
|
+
}
|
|
206
|
+
fills.push(formatting);
|
|
207
|
+
}
|
|
208
|
+
}
|
|
209
|
+
// Parse cellXfs (cell format definitions)
|
|
210
|
+
const cellXfsNode = (0, xmlUtils_1.getElementsByTagName)(xml, "cellXfs")[0];
|
|
211
|
+
if (cellXfsNode) {
|
|
212
|
+
const xfNodes = (0, xmlUtils_1.getElementsByTagName)(cellXfsNode, "xf");
|
|
213
|
+
for (let i = 0; i < xfNodes.length; i++) {
|
|
214
|
+
const xf = xfNodes[i];
|
|
215
|
+
const formatting = {};
|
|
216
|
+
const fontId = xf.getAttribute("fontId");
|
|
217
|
+
if (fontId) {
|
|
218
|
+
const fontIdx = parseInt(fontId);
|
|
219
|
+
if (fonts[fontIdx]) {
|
|
220
|
+
Object.assign(formatting, fonts[fontIdx]);
|
|
221
|
+
}
|
|
222
|
+
}
|
|
223
|
+
const fillId = xf.getAttribute("fillId");
|
|
224
|
+
if (fillId) {
|
|
225
|
+
const fillIdx = parseInt(fillId);
|
|
226
|
+
if (fills[fillIdx] && fills[fillIdx].backgroundColor) {
|
|
227
|
+
formatting.backgroundColor = fills[fillIdx].backgroundColor;
|
|
228
|
+
}
|
|
229
|
+
}
|
|
230
|
+
const alignmentNode = (0, xmlUtils_1.getElementsByTagName)(xf, "alignment")[0];
|
|
231
|
+
if (alignmentNode) {
|
|
232
|
+
const horizontal = alignmentNode.getAttribute("horizontal");
|
|
233
|
+
if (horizontal === 'center' || horizontal === 'right' || horizontal === 'justify' || horizontal === 'left') {
|
|
234
|
+
formatting.alignment = horizontal;
|
|
235
|
+
}
|
|
236
|
+
}
|
|
237
|
+
cellFormatMap[i] = formatting;
|
|
238
|
+
}
|
|
239
|
+
}
|
|
240
|
+
}
|
|
241
|
+
const attachments = [];
|
|
242
|
+
const mediaFiles = files.filter(f => f.path.match(/xl\/media\/.*/));
|
|
243
|
+
const chartFiles = files.filter(f => f.path.match(chartsRegex));
|
|
244
|
+
// Map to store image details by drawing file path and relationship ID
|
|
245
|
+
const drawingImageMap = {};
|
|
246
|
+
if (config.extractAttachments) {
|
|
247
|
+
// 1. Parse Drawing Rels to map rIds to media paths
|
|
248
|
+
const drawingRelsFiles = files.filter(f => f.path.match(drawingRelsRegex));
|
|
249
|
+
for (const relFile of drawingRelsFiles) {
|
|
250
|
+
const drawingFilename = relFile.path.split('/').pop()?.replace('.rels', '') || '';
|
|
251
|
+
const drawingPath = `xl/drawings/${drawingFilename}`;
|
|
252
|
+
const relsXml = (0, xmlUtils_1.parseXmlString)(relFile.content.toString());
|
|
253
|
+
const relationships = (0, xmlUtils_1.getElementsByTagName)(relsXml, "Relationship");
|
|
254
|
+
if (!drawingImageMap[drawingPath]) {
|
|
255
|
+
drawingImageMap[drawingPath] = {};
|
|
256
|
+
}
|
|
257
|
+
for (const rel of relationships) {
|
|
258
|
+
const id = rel.getAttribute("Id");
|
|
259
|
+
const target = rel.getAttribute("Target");
|
|
260
|
+
if (id && target && target.includes('media/')) {
|
|
261
|
+
// Target is usually like "../media/image1.png"
|
|
262
|
+
const mediaPath = 'xl/' + target.replace('../', '');
|
|
263
|
+
drawingImageMap[drawingPath][id] = { path: mediaPath };
|
|
264
|
+
}
|
|
265
|
+
}
|
|
266
|
+
}
|
|
267
|
+
// 2. Parse Drawings to get Alt Text and link to Rels
|
|
268
|
+
const drawingFiles = files.filter(f => f.path.match(drawingsRegex));
|
|
269
|
+
for (const drawingFile of drawingFiles) {
|
|
270
|
+
const xml = (0, xmlUtils_1.parseXmlString)(drawingFile.content.toString());
|
|
271
|
+
const pics = (0, xmlUtils_1.getElementsByTagName)(xml, "xdr:pic"); // SpreadsheetML drawing
|
|
272
|
+
const rels = drawingImageMap[drawingFile.path] || {};
|
|
273
|
+
for (const pic of pics) {
|
|
274
|
+
const blipFill = (0, xmlUtils_1.getElementsByTagName)(pic, "xdr:blipFill")[0];
|
|
275
|
+
const blip = blipFill ? (0, xmlUtils_1.getElementsByTagName)(blipFill, "a:blip")[0] : null;
|
|
276
|
+
const embedId = blip ? blip.getAttribute("r:embed") : null;
|
|
277
|
+
const nvPicPr = (0, xmlUtils_1.getElementsByTagName)(pic, "xdr:nvPicPr")[0];
|
|
278
|
+
const cNvPr = nvPicPr ? (0, xmlUtils_1.getElementsByTagName)(nvPicPr, "xdr:cNvPr")[0] : null;
|
|
279
|
+
const altText = cNvPr ? (cNvPr.getAttribute("descr") || cNvPr.getAttribute("name")) : undefined;
|
|
280
|
+
if (embedId && rels[embedId]) {
|
|
281
|
+
rels[embedId].altText = altText || '';
|
|
282
|
+
}
|
|
283
|
+
}
|
|
284
|
+
}
|
|
285
|
+
// 3. Process Media Files
|
|
286
|
+
for (const media of mediaFiles) {
|
|
287
|
+
const attachment = (0, imageUtils_1.createAttachment)(media.path.split('/').pop() || 'image', media.content);
|
|
288
|
+
// Try to find alt text for this media
|
|
289
|
+
let altText = '';
|
|
290
|
+
for (const drawingPath in drawingImageMap) {
|
|
291
|
+
for (const rId in drawingImageMap[drawingPath]) {
|
|
292
|
+
if (drawingImageMap[drawingPath][rId].path === media.path) {
|
|
293
|
+
altText = drawingImageMap[drawingPath][rId].altText || '';
|
|
294
|
+
break;
|
|
295
|
+
}
|
|
296
|
+
}
|
|
297
|
+
if (altText)
|
|
298
|
+
break;
|
|
299
|
+
}
|
|
300
|
+
if (altText)
|
|
301
|
+
attachment.altText = altText;
|
|
302
|
+
attachments.push(attachment);
|
|
303
|
+
if (config.ocr) {
|
|
304
|
+
if (attachment.mimeType.startsWith('image/')) {
|
|
305
|
+
try {
|
|
306
|
+
const ocrText = await (0, ocrUtils_1.performOcr)(media.content, config.ocrLanguage);
|
|
307
|
+
if (ocrText.trim()) {
|
|
308
|
+
attachment.ocrText = ocrText.trim();
|
|
309
|
+
}
|
|
310
|
+
}
|
|
311
|
+
catch (e) {
|
|
312
|
+
(0, errorUtils_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
|
|
313
|
+
}
|
|
314
|
+
}
|
|
315
|
+
}
|
|
316
|
+
}
|
|
317
|
+
for (const chart of chartFiles) {
|
|
318
|
+
const attachment = {
|
|
319
|
+
type: 'chart',
|
|
320
|
+
mimeType: 'application/vnd.openxmlformats-officedocument.wordprocessingml.document',
|
|
321
|
+
data: chart.content.toString('base64'),
|
|
322
|
+
name: chart.path.split('/').pop() || '',
|
|
323
|
+
extension: 'xml'
|
|
324
|
+
};
|
|
325
|
+
// Extract structured chart data
|
|
326
|
+
try {
|
|
327
|
+
const chartData = (0, chartUtils_1.extractChartData)(chart.content);
|
|
328
|
+
attachment.chartData = chartData;
|
|
329
|
+
}
|
|
330
|
+
catch (e) {
|
|
331
|
+
(0, errorUtils_1.logWarning)(`Failed to extract chart data from ${chart.path}:`, config, e);
|
|
332
|
+
}
|
|
333
|
+
attachments.push(attachment);
|
|
334
|
+
}
|
|
335
|
+
}
|
|
336
|
+
// Build map of drawing rId -> chart attachment name for linking
|
|
337
|
+
const drawingChartMap = {};
|
|
338
|
+
if (config.extractAttachments) {
|
|
339
|
+
const drawingRelsFiles = files.filter(f => f.path.match(drawingRelsRegex));
|
|
340
|
+
for (const relFile of drawingRelsFiles) {
|
|
341
|
+
const drawingFilename = relFile.path.split('/').pop()?.replace('.rels', '') || '';
|
|
342
|
+
const drawingPath = `xl/drawings/${drawingFilename}`;
|
|
343
|
+
const relsXml = (0, xmlUtils_1.parseXmlString)(relFile.content.toString());
|
|
344
|
+
const relationships = (0, xmlUtils_1.getElementsByTagName)(relsXml, "Relationship");
|
|
345
|
+
if (!drawingChartMap[drawingPath]) {
|
|
346
|
+
drawingChartMap[drawingPath] = {};
|
|
347
|
+
}
|
|
348
|
+
for (const rel of relationships) {
|
|
349
|
+
const id = rel.getAttribute("Id");
|
|
350
|
+
const target = rel.getAttribute("Target");
|
|
351
|
+
const type = rel.getAttribute("Type");
|
|
352
|
+
if (id && target && type && type.includes('chart')) {
|
|
353
|
+
// Target is like "../charts/chart1.xml"
|
|
354
|
+
const chartName = target.split('/').pop() || '';
|
|
355
|
+
drawingChartMap[drawingPath][id] = chartName;
|
|
356
|
+
}
|
|
357
|
+
}
|
|
358
|
+
}
|
|
359
|
+
}
|
|
360
|
+
// Parse workbook.xml to get sheet names and map them to sheet files
|
|
361
|
+
const sheetNameMap = {};
|
|
362
|
+
const workbookFile = files.find(f => f.path === 'xl/workbook.xml');
|
|
363
|
+
const workbookRelsFile = files.find(f => f.path === 'xl/_rels/workbook.xml.rels');
|
|
364
|
+
if (workbookFile && workbookRelsFile) {
|
|
365
|
+
// Parse rels to get rId -> file mapping
|
|
366
|
+
const relsXml = (0, xmlUtils_1.parseXmlString)(workbookRelsFile.content.toString());
|
|
367
|
+
const relationships = (0, xmlUtils_1.getElementsByTagName)(relsXml, "Relationship");
|
|
368
|
+
const rIdToFile = {};
|
|
369
|
+
for (const rel of relationships) {
|
|
370
|
+
const rId = rel.getAttribute("Id");
|
|
371
|
+
const target = rel.getAttribute("Target");
|
|
372
|
+
if (rId && target) {
|
|
373
|
+
// Target is like "worksheets/sheet1.xml"
|
|
374
|
+
const filename = target.split('/').pop() || '';
|
|
375
|
+
rIdToFile[rId] = filename;
|
|
376
|
+
}
|
|
377
|
+
}
|
|
378
|
+
// Parse workbook.xml to get sheet name -> rId mapping
|
|
379
|
+
const workbookXml = (0, xmlUtils_1.parseXmlString)(workbookFile.content.toString());
|
|
380
|
+
const sheets = (0, xmlUtils_1.getElementsByTagName)(workbookXml, "sheet");
|
|
381
|
+
for (const sheet of sheets) {
|
|
382
|
+
const name = sheet.getAttribute("name");
|
|
383
|
+
const rId = sheet.getAttribute("r:id");
|
|
384
|
+
if (name && rId && rIdToFile[rId]) {
|
|
385
|
+
sheetNameMap[rIdToFile[rId]] = name;
|
|
386
|
+
}
|
|
387
|
+
}
|
|
388
|
+
}
|
|
389
|
+
const content = [];
|
|
390
|
+
const rawContents = [];
|
|
391
|
+
for (const file of files) {
|
|
392
|
+
if (file.path.match(mediaFileRegex))
|
|
393
|
+
continue;
|
|
394
|
+
if (file.path === stringsFilePath)
|
|
395
|
+
continue;
|
|
396
|
+
if (file.path === 'xl/styles.xml')
|
|
397
|
+
continue;
|
|
398
|
+
if (file.path.match(drawingsRegex))
|
|
399
|
+
continue;
|
|
400
|
+
if (file.path.match(chartsRegex))
|
|
401
|
+
continue;
|
|
402
|
+
if (file.path.match(relsRegex))
|
|
403
|
+
continue;
|
|
404
|
+
if (file.path.match(drawingRelsRegex))
|
|
405
|
+
continue;
|
|
406
|
+
if (file.path.match(sheetsRegex)) {
|
|
407
|
+
if (config.includeRawContent) {
|
|
408
|
+
rawContents.push(file.content.toString());
|
|
409
|
+
}
|
|
410
|
+
const rows = [];
|
|
411
|
+
const rowRegex = /<row.*?>[\s\S]*?<\/row>/g;
|
|
412
|
+
const rowMatches = file.content.toString().match(rowRegex);
|
|
413
|
+
if (rowMatches) {
|
|
414
|
+
for (const rowXml of rowMatches) {
|
|
415
|
+
const cells = [];
|
|
416
|
+
const cRegex = /<c.*?>[\s\S]*?<\/c>/g;
|
|
417
|
+
const cMatches = rowXml.match(cRegex);
|
|
418
|
+
const rMatch = rowXml.match(/r="(\d+)"/);
|
|
419
|
+
const rowIndex = rMatch ? parseInt(rMatch[1]) - 1 : 0;
|
|
420
|
+
if (cMatches) {
|
|
421
|
+
for (const cXml of cMatches) {
|
|
422
|
+
// Extract cell value
|
|
423
|
+
const typeMatch = cXml.match(/t="([a-z]+)"/);
|
|
424
|
+
const type = typeMatch ? typeMatch[1] : 'n'; // n = number (default)
|
|
425
|
+
const vMatch = cXml.match(/<v>(.*?)<\/v>/);
|
|
426
|
+
const tMatch = cXml.match(/<t>(.*?)<\/t>/);
|
|
427
|
+
let text = '';
|
|
428
|
+
let cellNodes = [];
|
|
429
|
+
if (type === 's' && vMatch) {
|
|
430
|
+
const idx = parseInt(vMatch[1]);
|
|
431
|
+
const content = sharedStrings[idx];
|
|
432
|
+
if (Array.isArray(content)) {
|
|
433
|
+
// Rich text runs
|
|
434
|
+
// Deep copy runs to avoid reference issues if reused
|
|
435
|
+
cellNodes = JSON.parse(JSON.stringify(content));
|
|
436
|
+
text = cellNodes.map(n => n.text).join('');
|
|
437
|
+
}
|
|
438
|
+
else {
|
|
439
|
+
text = content || '';
|
|
440
|
+
}
|
|
441
|
+
}
|
|
442
|
+
else if (type === 'inlineStr' && tMatch) {
|
|
443
|
+
text = tMatch[1];
|
|
444
|
+
}
|
|
445
|
+
else if (vMatch) {
|
|
446
|
+
text = vMatch[1];
|
|
447
|
+
}
|
|
448
|
+
// Parse cell coordinate
|
|
449
|
+
const coordMatch = cXml.match(/r="([A-Z]+)(\d+)"/);
|
|
450
|
+
const colStr = coordMatch ? coordMatch[1] : '';
|
|
451
|
+
const colIndex = colStr.charCodeAt(0) - 'A'.charCodeAt(0);
|
|
452
|
+
if (text || cellNodes.length > 0) {
|
|
453
|
+
// Extract cell style index
|
|
454
|
+
const styleMatch = cXml.match(/s="(\d+)"/);
|
|
455
|
+
const styleIdx = styleMatch ? parseInt(styleMatch[1]) : undefined;
|
|
456
|
+
const cellFormatting = (styleIdx !== undefined && cellFormatMap[styleIdx]) ? cellFormatMap[styleIdx] : {};
|
|
457
|
+
if (cellNodes.length > 0) {
|
|
458
|
+
// If we have specific runs, merge cell styles into them if run style is missing
|
|
459
|
+
// But usually run style overrides cell style (except maybe background)
|
|
460
|
+
for (const node of cellNodes) {
|
|
461
|
+
if (!node.formatting)
|
|
462
|
+
node.formatting = {};
|
|
463
|
+
// Cell background always applies
|
|
464
|
+
if (cellFormatting.backgroundColor)
|
|
465
|
+
node.formatting.backgroundColor = cellFormatting.backgroundColor;
|
|
466
|
+
// Cell alignment always applies
|
|
467
|
+
if (cellFormatting.alignment)
|
|
468
|
+
node.formatting.alignment = cellFormatting.alignment;
|
|
469
|
+
// Font defaults from cell style if not in run
|
|
470
|
+
if (!node.formatting.font && cellFormatting.font)
|
|
471
|
+
node.formatting.font = cellFormatting.font;
|
|
472
|
+
if (!node.formatting.size && cellFormatting.size)
|
|
473
|
+
node.formatting.size = cellFormatting.size;
|
|
474
|
+
}
|
|
475
|
+
}
|
|
476
|
+
else {
|
|
477
|
+
// Simple text node
|
|
478
|
+
cellNodes.push({
|
|
479
|
+
type: 'text',
|
|
480
|
+
text: text,
|
|
481
|
+
formatting: cellFormatting
|
|
482
|
+
});
|
|
483
|
+
}
|
|
484
|
+
const cellNode = {
|
|
485
|
+
type: 'cell',
|
|
486
|
+
text: text,
|
|
487
|
+
children: cellNodes,
|
|
488
|
+
metadata: { row: rowIndex, col: colIndex }
|
|
489
|
+
};
|
|
490
|
+
if (config.includeRawContent) {
|
|
491
|
+
cellNode.rawContent = cXml;
|
|
492
|
+
}
|
|
493
|
+
cells.push(cellNode);
|
|
494
|
+
}
|
|
495
|
+
}
|
|
496
|
+
}
|
|
497
|
+
if (cells.length > 0) {
|
|
498
|
+
const rowNode = {
|
|
499
|
+
type: 'row',
|
|
500
|
+
children: cells,
|
|
501
|
+
metadata: undefined
|
|
502
|
+
};
|
|
503
|
+
if (config.includeRawContent) {
|
|
504
|
+
rowNode.rawContent = rowXml;
|
|
505
|
+
}
|
|
506
|
+
rows.push(rowNode);
|
|
507
|
+
}
|
|
508
|
+
}
|
|
509
|
+
}
|
|
510
|
+
// Handle Drawings in Sheet (images and charts)
|
|
511
|
+
if (config.extractAttachments) {
|
|
512
|
+
// Parse Sheet Rels to map drawing rIds
|
|
513
|
+
const sheetFilename = file.path.split('/').pop() || '';
|
|
514
|
+
const relsFilename = `xl/worksheets/_rels/${sheetFilename}.rels`;
|
|
515
|
+
const relsFile = files.find(f => f.path === relsFilename);
|
|
516
|
+
const drawingMap = {}; // rId -> drawingPath
|
|
517
|
+
if (relsFile) {
|
|
518
|
+
const relsXml = (0, xmlUtils_1.parseXmlString)(relsFile.content.toString());
|
|
519
|
+
const relationships = (0, xmlUtils_1.getElementsByTagName)(relsXml, "Relationship");
|
|
520
|
+
for (const rel of relationships) {
|
|
521
|
+
const id = rel.getAttribute("Id");
|
|
522
|
+
const target = rel.getAttribute("Target");
|
|
523
|
+
const type = rel.getAttribute("Type");
|
|
524
|
+
if (id && target && type && type.includes('drawing')) {
|
|
525
|
+
drawingMap[id] = 'xl/drawings/' + target.replace('../drawings/', '');
|
|
526
|
+
}
|
|
527
|
+
}
|
|
528
|
+
}
|
|
529
|
+
const drawingMatches = file.content.toString().match(/<drawing r:id="(.*?)"/g);
|
|
530
|
+
if (drawingMatches) {
|
|
531
|
+
for (const match of drawingMatches) {
|
|
532
|
+
const rIdMatch = match.match(/r:id="(.*?)"/);
|
|
533
|
+
const rId = rIdMatch ? rIdMatch[1] : null;
|
|
534
|
+
if (rId && drawingMap[rId]) {
|
|
535
|
+
const drawingPath = drawingMap[rId];
|
|
536
|
+
// Find all images in this drawing
|
|
537
|
+
const images = drawingImageMap[drawingPath];
|
|
538
|
+
if (images) {
|
|
539
|
+
for (const imgId in images) {
|
|
540
|
+
const imgInfo = images[imgId];
|
|
541
|
+
const attachment = attachments.find(a => a.name === imgInfo.path.split('/').pop());
|
|
542
|
+
if (attachment) {
|
|
543
|
+
const imageNode = {
|
|
544
|
+
type: 'image',
|
|
545
|
+
text: '',
|
|
546
|
+
children: [],
|
|
547
|
+
metadata: {
|
|
548
|
+
attachmentName: attachment.name || 'unknown',
|
|
549
|
+
altText: imgInfo.altText || undefined
|
|
550
|
+
}
|
|
551
|
+
};
|
|
552
|
+
rows.push(imageNode);
|
|
553
|
+
}
|
|
554
|
+
}
|
|
555
|
+
}
|
|
556
|
+
// Find all charts in this drawing
|
|
557
|
+
const charts = drawingChartMap[drawingPath];
|
|
558
|
+
if (charts) {
|
|
559
|
+
for (const chartRId in charts) {
|
|
560
|
+
const chartName = charts[chartRId];
|
|
561
|
+
const attachment = attachments.find(a => a.name === chartName);
|
|
562
|
+
if (attachment) {
|
|
563
|
+
const chartNode = {
|
|
564
|
+
type: 'chart',
|
|
565
|
+
text: '',
|
|
566
|
+
children: [],
|
|
567
|
+
metadata: {
|
|
568
|
+
attachmentName: chartName
|
|
569
|
+
}
|
|
570
|
+
};
|
|
571
|
+
rows.push(chartNode);
|
|
572
|
+
}
|
|
573
|
+
}
|
|
574
|
+
}
|
|
575
|
+
}
|
|
576
|
+
}
|
|
577
|
+
}
|
|
578
|
+
}
|
|
579
|
+
// Get proper sheet name from workbook.xml mapping, fallback to filename
|
|
580
|
+
const sheetFileName = file.path.split('/').pop() || 'Sheet';
|
|
581
|
+
const sheetName = sheetNameMap[sheetFileName] || sheetFileName;
|
|
582
|
+
content.push({
|
|
583
|
+
type: 'sheet',
|
|
584
|
+
children: rows,
|
|
585
|
+
metadata: { sheetName },
|
|
586
|
+
rawContent: config.includeRawContent ? file.content.toString() : undefined
|
|
587
|
+
});
|
|
588
|
+
}
|
|
589
|
+
}
|
|
590
|
+
const corePropsFile = files.find(f => f.path.match(corePropsFileRegex));
|
|
591
|
+
const metadata = corePropsFile ? (0, xmlUtils_1.parseOfficeMetadata)(corePropsFile.content.toString()) : {};
|
|
592
|
+
// Link OCR text and chart data to content nodes (like PPTX parser)
|
|
593
|
+
const assignAttachmentData = (nodes) => {
|
|
594
|
+
for (const node of nodes) {
|
|
595
|
+
if ('attachmentName' in (node.metadata || {})) {
|
|
596
|
+
const meta = node.metadata;
|
|
597
|
+
const attachment = attachments.find(a => a.name === meta.attachmentName);
|
|
598
|
+
if (attachment) {
|
|
599
|
+
if (node.type === 'image') {
|
|
600
|
+
// Link OCR text to image node
|
|
601
|
+
if (attachment.ocrText) {
|
|
602
|
+
node.text = attachment.ocrText;
|
|
603
|
+
}
|
|
604
|
+
// Copy altText to attachment
|
|
605
|
+
if (meta.altText) {
|
|
606
|
+
attachment.altText = meta.altText;
|
|
607
|
+
}
|
|
608
|
+
}
|
|
609
|
+
if (node.type === 'chart') {
|
|
610
|
+
// Link chart data text to chart node
|
|
611
|
+
if (attachment.chartData) {
|
|
612
|
+
node.text = attachment.chartData.rawTexts.join(config.newlineDelimiter || '\n');
|
|
613
|
+
}
|
|
614
|
+
}
|
|
615
|
+
}
|
|
616
|
+
}
|
|
617
|
+
if (node.children) {
|
|
618
|
+
assignAttachmentData(node.children);
|
|
619
|
+
}
|
|
620
|
+
}
|
|
621
|
+
};
|
|
622
|
+
assignAttachmentData(content);
|
|
623
|
+
return {
|
|
624
|
+
type: 'xlsx',
|
|
625
|
+
metadata: metadata,
|
|
626
|
+
content: content,
|
|
627
|
+
attachments: attachments,
|
|
628
|
+
toText: () => content.map(c => {
|
|
629
|
+
// Recursive text extraction
|
|
630
|
+
const getText = (node) => {
|
|
631
|
+
let t = '';
|
|
632
|
+
if (node.children) {
|
|
633
|
+
t += node.children.map(getText).filter(t => t != '').join(!node.children[0]?.children ? '' : config.newlineDelimiter ?? '\n');
|
|
634
|
+
}
|
|
635
|
+
else
|
|
636
|
+
t += node.text || '';
|
|
637
|
+
return t;
|
|
638
|
+
};
|
|
639
|
+
return getText(c);
|
|
640
|
+
}).filter(t => t != '').join(config.newlineDelimiter ?? '\n')
|
|
641
|
+
};
|
|
642
|
+
};
|
|
643
|
+
exports.parseExcel = parseExcel;
|