officeparser 6.0.7 → 6.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +136 -52
- package/dist/OfficeParser.d.ts +10 -1
- package/dist/OfficeParser.js +44 -56
- package/dist/cli.d.ts +20 -0
- package/dist/cli.js +117 -0
- package/dist/index.d.ts +4 -4
- package/dist/index.js +7 -59
- package/dist/index.mjs +18 -0
- package/dist/officeparser.browser.d.ts +133 -3
- package/dist/officeparser.browser.iife.js +115 -0
- package/dist/officeparser.browser.mjs +114 -0
- package/dist/parsers/ExcelParser.d.ts +1 -1
- package/dist/parsers/ExcelParser.js +76 -68
- package/dist/parsers/OpenOfficeParser.d.ts +1 -1
- package/dist/parsers/OpenOfficeParser.js +224 -159
- package/dist/parsers/PdfParser.d.ts +1 -1
- package/dist/parsers/PdfParser.js +98 -94
- package/dist/parsers/PowerPointParser.d.ts +1 -1
- package/dist/parsers/PowerPointParser.js +188 -179
- package/dist/parsers/RtfParser.d.ts +21 -1
- package/dist/parsers/RtfParser.js +117 -48
- package/dist/parsers/WordParser.d.ts +2 -1
- package/dist/parsers/WordParser.js +214 -123
- package/dist/sbom.cdx.json +1807 -0
- package/dist/types.d.ts +123 -3
- package/dist/utils/chartUtils.js +2 -0
- package/dist/utils/dateUtils.d.ts +17 -0
- package/dist/utils/dateUtils.js +69 -0
- package/dist/utils/envUtils.d.ts +24 -0
- package/dist/utils/envUtils.js +69 -0
- package/dist/utils/moduleLoader.d.ts +2 -1
- package/dist/utils/moduleLoader.js +9 -39
- package/dist/utils/ocrUtils.d.ts +16 -12
- package/dist/utils/ocrUtils.js +186 -25
- package/dist/utils/xmlUtils.d.ts +80 -9
- package/dist/utils/xmlUtils.js +236 -18
- package/dist/utils/zipUtils.js +6 -47
- package/package.json +31 -16
- package/dist/officeParserBundle@6.0.7.js +0 -154
- package/dist/officeparser.browser.js +0 -154
|
@@ -21,7 +21,7 @@
|
|
|
21
21
|
* @module ExcelParser
|
|
22
22
|
* @see https://www.ecma-international.org/publications-and-standards/standards/ecma-376/
|
|
23
23
|
*/
|
|
24
|
-
import { OfficeParserAST, OfficeParserConfig } from '../types';
|
|
24
|
+
import { OfficeParserAST, OfficeParserConfig } from '../types.js';
|
|
25
25
|
/**
|
|
26
26
|
* Parses an Excel spreadsheet (.xlsx) and extracts sheets, rows, and cells.
|
|
27
27
|
*
|
|
@@ -24,12 +24,12 @@
|
|
|
24
24
|
*/
|
|
25
25
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
26
26
|
exports.parseExcel = void 0;
|
|
27
|
-
const
|
|
28
|
-
const
|
|
29
|
-
const
|
|
30
|
-
const
|
|
31
|
-
const
|
|
32
|
-
const
|
|
27
|
+
const chartUtils_js_1 = require("../utils/chartUtils.js");
|
|
28
|
+
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
29
|
+
const imageUtils_js_1 = require("../utils/imageUtils.js");
|
|
30
|
+
const ocrUtils_js_1 = require("../utils/ocrUtils.js");
|
|
31
|
+
const xmlUtils_js_1 = require("../utils/xmlUtils.js");
|
|
32
|
+
const zipUtils_js_1 = require("../utils/zipUtils.js");
|
|
33
33
|
/**
|
|
34
34
|
* Parses an Excel spreadsheet (.xlsx) and extracts sheets, rows, and cells.
|
|
35
35
|
*
|
|
@@ -44,9 +44,10 @@ const parseExcel = async (buffer, config) => {
|
|
|
44
44
|
const stringsFilePath = 'xl/sharedStrings.xml';
|
|
45
45
|
const mediaFileRegex = /xl\/media\/.*/;
|
|
46
46
|
const corePropsFileRegex = /docProps\/core\.xml/;
|
|
47
|
+
const customPropsFileRegex = /docProps\/custom\.xml/;
|
|
47
48
|
const relsRegex = /xl\/worksheets\/_rels\/sheet\d+\.xml\.rels/g;
|
|
48
49
|
const drawingRelsRegex = /xl\/drawings\/_rels\/drawing\d+\.xml\.rels/g;
|
|
49
|
-
const files = await (0,
|
|
50
|
+
const files = await (0, zipUtils_js_1.extractFiles)(buffer, (x) => !!x.match(sheetsRegex) ||
|
|
50
51
|
!!x.match(drawingsRegex) ||
|
|
51
52
|
!!x.match(chartsRegex) ||
|
|
52
53
|
x === stringsFilePath ||
|
|
@@ -54,47 +55,48 @@ const parseExcel = async (buffer, config) => {
|
|
|
54
55
|
x === 'xl/workbook.xml' ||
|
|
55
56
|
x === 'xl/_rels/workbook.xml.rels' ||
|
|
56
57
|
!!x.match(corePropsFileRegex) ||
|
|
58
|
+
!!x.match(customPropsFileRegex) ||
|
|
57
59
|
(!!config.extractAttachments && (!!x.match(mediaFileRegex) || !!x.match(relsRegex) || !!x.match(drawingRelsRegex))));
|
|
58
60
|
const sharedStringsFile = files.find(f => f.path === stringsFilePath);
|
|
59
61
|
// Updated to store structured content (rich text runs) or simple string
|
|
60
62
|
const sharedStrings = [];
|
|
61
63
|
if (sharedStringsFile) {
|
|
62
|
-
const xml = (0,
|
|
63
|
-
const siNodes = (0,
|
|
64
|
+
const xml = (0, xmlUtils_js_1.parseXmlString)(sharedStringsFile.content.toString());
|
|
65
|
+
const siNodes = (0, xmlUtils_js_1.getElementsByTagName)(xml, "si");
|
|
64
66
|
for (const si of siNodes) {
|
|
65
|
-
const runNodes = (0,
|
|
67
|
+
const runNodes = (0, xmlUtils_js_1.getElementsByTagName)(si, "r");
|
|
66
68
|
if (runNodes.length > 0) {
|
|
67
69
|
// Rich text with runs
|
|
68
70
|
const runs = [];
|
|
69
71
|
for (const run of runNodes) {
|
|
70
|
-
const tNode = (0,
|
|
72
|
+
const tNode = (0, xmlUtils_js_1.getElementsByTagName)(run, "t")[0];
|
|
71
73
|
if (tNode) {
|
|
72
74
|
const text = tNode.textContent || '';
|
|
73
75
|
// Extract run formatting
|
|
74
|
-
const rPr = (0,
|
|
76
|
+
const rPr = (0, xmlUtils_js_1.getElementsByTagName)(run, "rPr")[0];
|
|
75
77
|
const formatting = {};
|
|
76
78
|
if (rPr) {
|
|
77
|
-
if ((0,
|
|
79
|
+
if ((0, xmlUtils_js_1.getElementsByTagName)(rPr, "b").length > 0)
|
|
78
80
|
formatting.bold = true;
|
|
79
|
-
if ((0,
|
|
81
|
+
if ((0, xmlUtils_js_1.getElementsByTagName)(rPr, "i").length > 0)
|
|
80
82
|
formatting.italic = true;
|
|
81
|
-
if ((0,
|
|
83
|
+
if ((0, xmlUtils_js_1.getElementsByTagName)(rPr, "u").length > 0)
|
|
82
84
|
formatting.underline = true;
|
|
83
|
-
if ((0,
|
|
85
|
+
if ((0, xmlUtils_js_1.getElementsByTagName)(rPr, "strike").length > 0)
|
|
84
86
|
formatting.strikethrough = true;
|
|
85
|
-
const sz = (0,
|
|
87
|
+
const sz = (0, xmlUtils_js_1.getElementsByTagName)(rPr, "sz")[0];
|
|
86
88
|
if (sz)
|
|
87
89
|
formatting.size = sz.getAttribute("val") + 'pt';
|
|
88
|
-
const color = (0,
|
|
90
|
+
const color = (0, xmlUtils_js_1.getElementsByTagName)(rPr, "color")[0];
|
|
89
91
|
if (color) {
|
|
90
92
|
const rgb = color.getAttribute("rgb");
|
|
91
93
|
if (rgb)
|
|
92
94
|
formatting.color = '#' + rgb.substring(2);
|
|
93
95
|
}
|
|
94
|
-
const rFont = (0,
|
|
96
|
+
const rFont = (0, xmlUtils_js_1.getElementsByTagName)(rPr, "rFont")[0];
|
|
95
97
|
if (rFont)
|
|
96
98
|
formatting.font = rFont.getAttribute("val") || undefined;
|
|
97
|
-
const vertAlign = (0,
|
|
99
|
+
const vertAlign = (0, xmlUtils_js_1.getElementsByTagName)(rPr, "vertAlign")[0];
|
|
98
100
|
if (vertAlign) {
|
|
99
101
|
const val = vertAlign.getAttribute("val");
|
|
100
102
|
if (val === "subscript")
|
|
@@ -114,7 +116,7 @@ const parseExcel = async (buffer, config) => {
|
|
|
114
116
|
}
|
|
115
117
|
else {
|
|
116
118
|
// Simple text case
|
|
117
|
-
const tNodes = (0,
|
|
119
|
+
const tNodes = (0, xmlUtils_js_1.getElementsByTagName)(si, "t");
|
|
118
120
|
let text = '';
|
|
119
121
|
for (const t of tNodes) {
|
|
120
122
|
text += t.textContent || '';
|
|
@@ -127,41 +129,41 @@ const parseExcel = async (buffer, config) => {
|
|
|
127
129
|
const stylesFile = files.find(f => f.path === 'xl/styles.xml');
|
|
128
130
|
const cellFormatMap = {};
|
|
129
131
|
if (stylesFile) {
|
|
130
|
-
const xml = (0,
|
|
132
|
+
const xml = (0, xmlUtils_js_1.parseXmlString)(stylesFile.content.toString());
|
|
131
133
|
// Parse fonts
|
|
132
|
-
const fontsNode = (0,
|
|
134
|
+
const fontsNode = (0, xmlUtils_js_1.getElementsByTagName)(xml, "fonts")[0];
|
|
133
135
|
const fonts = [];
|
|
134
136
|
if (fontsNode) {
|
|
135
|
-
const fontNodes = (0,
|
|
137
|
+
const fontNodes = (0, xmlUtils_js_1.getElementsByTagName)(fontsNode, "font");
|
|
136
138
|
for (const font of fontNodes) {
|
|
137
139
|
const formatting = {};
|
|
138
|
-
if ((0,
|
|
140
|
+
if ((0, xmlUtils_js_1.getElementsByTagName)(font, "b").length > 0)
|
|
139
141
|
formatting.bold = true;
|
|
140
|
-
if ((0,
|
|
142
|
+
if ((0, xmlUtils_js_1.getElementsByTagName)(font, "i").length > 0)
|
|
141
143
|
formatting.italic = true;
|
|
142
|
-
if ((0,
|
|
144
|
+
if ((0, xmlUtils_js_1.getElementsByTagName)(font, "u").length > 0)
|
|
143
145
|
formatting.underline = true;
|
|
144
|
-
if ((0,
|
|
146
|
+
if ((0, xmlUtils_js_1.getElementsByTagName)(font, "strike").length > 0)
|
|
145
147
|
formatting.strikethrough = true;
|
|
146
|
-
const szNode = (0,
|
|
148
|
+
const szNode = (0, xmlUtils_js_1.getElementsByTagName)(font, "sz")[0];
|
|
147
149
|
if (szNode) {
|
|
148
150
|
const val = szNode.getAttribute("val");
|
|
149
151
|
if (val)
|
|
150
152
|
formatting.size = val + 'pt';
|
|
151
153
|
}
|
|
152
|
-
const colorNode = (0,
|
|
154
|
+
const colorNode = (0, xmlUtils_js_1.getElementsByTagName)(font, "color")[0];
|
|
153
155
|
if (colorNode) {
|
|
154
156
|
const rgb = colorNode.getAttribute("rgb");
|
|
155
157
|
if (rgb)
|
|
156
158
|
formatting.color = '#' + rgb.substring(2); // Remove alpha channel
|
|
157
159
|
}
|
|
158
|
-
const nameNode = (0,
|
|
160
|
+
const nameNode = (0, xmlUtils_js_1.getElementsByTagName)(font, "name")[0];
|
|
159
161
|
if (nameNode) {
|
|
160
162
|
const val = nameNode.getAttribute("val");
|
|
161
163
|
if (val)
|
|
162
164
|
formatting.font = val;
|
|
163
165
|
}
|
|
164
|
-
const vertAlignNode = (0,
|
|
166
|
+
const vertAlignNode = (0, xmlUtils_js_1.getElementsByTagName)(font, "vertAlign")[0];
|
|
165
167
|
if (vertAlignNode) {
|
|
166
168
|
const val = vertAlignNode.getAttribute("val");
|
|
167
169
|
if (val === "subscript")
|
|
@@ -173,15 +175,15 @@ const parseExcel = async (buffer, config) => {
|
|
|
173
175
|
}
|
|
174
176
|
}
|
|
175
177
|
// Parse fills (for background color)
|
|
176
|
-
const fillsNode = (0,
|
|
178
|
+
const fillsNode = (0, xmlUtils_js_1.getElementsByTagName)(xml, "fills")[0];
|
|
177
179
|
const fills = [];
|
|
178
180
|
if (fillsNode) {
|
|
179
|
-
const fillNodes = (0,
|
|
181
|
+
const fillNodes = (0, xmlUtils_js_1.getElementsByTagName)(fillsNode, "fill");
|
|
180
182
|
for (const fill of fillNodes) {
|
|
181
183
|
const formatting = {};
|
|
182
|
-
const patternFill = (0,
|
|
184
|
+
const patternFill = (0, xmlUtils_js_1.getElementsByTagName)(fill, "patternFill")[0];
|
|
183
185
|
if (patternFill) {
|
|
184
|
-
const fgColor = (0,
|
|
186
|
+
const fgColor = (0, xmlUtils_js_1.getElementsByTagName)(patternFill, "fgColor")[0];
|
|
185
187
|
if (fgColor) {
|
|
186
188
|
const rgb = fgColor.getAttribute("rgb");
|
|
187
189
|
const theme = fgColor.getAttribute("theme");
|
|
@@ -207,9 +209,9 @@ const parseExcel = async (buffer, config) => {
|
|
|
207
209
|
}
|
|
208
210
|
}
|
|
209
211
|
// Parse cellXfs (cell format definitions)
|
|
210
|
-
const cellXfsNode = (0,
|
|
212
|
+
const cellXfsNode = (0, xmlUtils_js_1.getElementsByTagName)(xml, "cellXfs")[0];
|
|
211
213
|
if (cellXfsNode) {
|
|
212
|
-
const xfNodes = (0,
|
|
214
|
+
const xfNodes = (0, xmlUtils_js_1.getElementsByTagName)(cellXfsNode, "xf");
|
|
213
215
|
for (let i = 0; i < xfNodes.length; i++) {
|
|
214
216
|
const xf = xfNodes[i];
|
|
215
217
|
const formatting = {};
|
|
@@ -227,7 +229,7 @@ const parseExcel = async (buffer, config) => {
|
|
|
227
229
|
formatting.backgroundColor = fills[fillIdx].backgroundColor;
|
|
228
230
|
}
|
|
229
231
|
}
|
|
230
|
-
const alignmentNode = (0,
|
|
232
|
+
const alignmentNode = (0, xmlUtils_js_1.getElementsByTagName)(xf, "alignment")[0];
|
|
231
233
|
if (alignmentNode) {
|
|
232
234
|
const horizontal = alignmentNode.getAttribute("horizontal");
|
|
233
235
|
if (horizontal === 'center' || horizontal === 'right' || horizontal === 'justify' || horizontal === 'left') {
|
|
@@ -249,8 +251,8 @@ const parseExcel = async (buffer, config) => {
|
|
|
249
251
|
for (const relFile of drawingRelsFiles) {
|
|
250
252
|
const drawingFilename = relFile.path.split('/').pop()?.replace('.rels', '') || '';
|
|
251
253
|
const drawingPath = `xl/drawings/${drawingFilename}`;
|
|
252
|
-
const relsXml = (0,
|
|
253
|
-
const relationships = (0,
|
|
254
|
+
const relsXml = (0, xmlUtils_js_1.parseXmlString)(relFile.content.toString());
|
|
255
|
+
const relationships = (0, xmlUtils_js_1.getElementsByTagName)(relsXml, "Relationship");
|
|
254
256
|
if (!drawingImageMap[drawingPath]) {
|
|
255
257
|
drawingImageMap[drawingPath] = {};
|
|
256
258
|
}
|
|
@@ -267,15 +269,15 @@ const parseExcel = async (buffer, config) => {
|
|
|
267
269
|
// 2. Parse Drawings to get Alt Text and link to Rels
|
|
268
270
|
const drawingFiles = files.filter(f => f.path.match(drawingsRegex));
|
|
269
271
|
for (const drawingFile of drawingFiles) {
|
|
270
|
-
const xml = (0,
|
|
271
|
-
const pics = (0,
|
|
272
|
+
const xml = (0, xmlUtils_js_1.parseXmlString)(drawingFile.content.toString());
|
|
273
|
+
const pics = (0, xmlUtils_js_1.getElementsByTagName)(xml, "xdr:pic"); // SpreadsheetML drawing
|
|
272
274
|
const rels = drawingImageMap[drawingFile.path] || {};
|
|
273
275
|
for (const pic of pics) {
|
|
274
|
-
const blipFill = (0,
|
|
275
|
-
const blip = blipFill ? (0,
|
|
276
|
+
const blipFill = (0, xmlUtils_js_1.getElementsByTagName)(pic, "xdr:blipFill")[0];
|
|
277
|
+
const blip = blipFill ? (0, xmlUtils_js_1.getElementsByTagName)(blipFill, "a:blip")[0] : null;
|
|
276
278
|
const embedId = blip ? blip.getAttribute("r:embed") : null;
|
|
277
|
-
const nvPicPr = (0,
|
|
278
|
-
const cNvPr = nvPicPr ? (0,
|
|
279
|
+
const nvPicPr = (0, xmlUtils_js_1.getElementsByTagName)(pic, "xdr:nvPicPr")[0];
|
|
280
|
+
const cNvPr = nvPicPr ? (0, xmlUtils_js_1.getElementsByTagName)(nvPicPr, "xdr:cNvPr")[0] : null;
|
|
279
281
|
const altText = cNvPr ? (cNvPr.getAttribute("descr") || cNvPr.getAttribute("name")) : undefined;
|
|
280
282
|
if (embedId && rels[embedId]) {
|
|
281
283
|
rels[embedId].altText = altText || '';
|
|
@@ -284,7 +286,7 @@ const parseExcel = async (buffer, config) => {
|
|
|
284
286
|
}
|
|
285
287
|
// 3. Process Media Files
|
|
286
288
|
for (const media of mediaFiles) {
|
|
287
|
-
const attachment = (0,
|
|
289
|
+
const attachment = (0, imageUtils_js_1.createAttachment)(media.path.split('/').pop() || 'image', media.content);
|
|
288
290
|
// Try to find alt text for this media
|
|
289
291
|
let altText = '';
|
|
290
292
|
for (const drawingPath in drawingImageMap) {
|
|
@@ -303,13 +305,13 @@ const parseExcel = async (buffer, config) => {
|
|
|
303
305
|
if (config.ocr) {
|
|
304
306
|
if (attachment.mimeType.startsWith('image/')) {
|
|
305
307
|
try {
|
|
306
|
-
const ocrText = await (0,
|
|
307
|
-
if (ocrText
|
|
308
|
-
attachment.ocrText = ocrText
|
|
308
|
+
const ocrText = (await (0, ocrUtils_js_1.performOcr)(media.content, { language: config.ocrLanguage, ...config.ocrConfig })).trim();
|
|
309
|
+
if (ocrText) {
|
|
310
|
+
attachment.ocrText = ocrText;
|
|
309
311
|
}
|
|
310
312
|
}
|
|
311
313
|
catch (e) {
|
|
312
|
-
(0,
|
|
314
|
+
(0, errorUtils_js_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
|
|
313
315
|
}
|
|
314
316
|
}
|
|
315
317
|
}
|
|
@@ -324,11 +326,11 @@ const parseExcel = async (buffer, config) => {
|
|
|
324
326
|
};
|
|
325
327
|
// Extract structured chart data
|
|
326
328
|
try {
|
|
327
|
-
const chartData = (0,
|
|
329
|
+
const chartData = (0, chartUtils_js_1.extractChartData)(chart.content);
|
|
328
330
|
attachment.chartData = chartData;
|
|
329
331
|
}
|
|
330
332
|
catch (e) {
|
|
331
|
-
(0,
|
|
333
|
+
(0, errorUtils_js_1.logWarning)(`Failed to extract chart data from ${chart.path}:`, config, e);
|
|
332
334
|
}
|
|
333
335
|
attachments.push(attachment);
|
|
334
336
|
}
|
|
@@ -340,8 +342,8 @@ const parseExcel = async (buffer, config) => {
|
|
|
340
342
|
for (const relFile of drawingRelsFiles) {
|
|
341
343
|
const drawingFilename = relFile.path.split('/').pop()?.replace('.rels', '') || '';
|
|
342
344
|
const drawingPath = `xl/drawings/${drawingFilename}`;
|
|
343
|
-
const relsXml = (0,
|
|
344
|
-
const relationships = (0,
|
|
345
|
+
const relsXml = (0, xmlUtils_js_1.parseXmlString)(relFile.content.toString());
|
|
346
|
+
const relationships = (0, xmlUtils_js_1.getElementsByTagName)(relsXml, "Relationship");
|
|
345
347
|
if (!drawingChartMap[drawingPath]) {
|
|
346
348
|
drawingChartMap[drawingPath] = {};
|
|
347
349
|
}
|
|
@@ -363,8 +365,8 @@ const parseExcel = async (buffer, config) => {
|
|
|
363
365
|
const workbookRelsFile = files.find(f => f.path === 'xl/_rels/workbook.xml.rels');
|
|
364
366
|
if (workbookFile && workbookRelsFile) {
|
|
365
367
|
// Parse rels to get rId -> file mapping
|
|
366
|
-
const relsXml = (0,
|
|
367
|
-
const relationships = (0,
|
|
368
|
+
const relsXml = (0, xmlUtils_js_1.parseXmlString)(workbookRelsFile.content.toString());
|
|
369
|
+
const relationships = (0, xmlUtils_js_1.getElementsByTagName)(relsXml, "Relationship");
|
|
368
370
|
const rIdToFile = {};
|
|
369
371
|
for (const rel of relationships) {
|
|
370
372
|
const rId = rel.getAttribute("Id");
|
|
@@ -376,8 +378,8 @@ const parseExcel = async (buffer, config) => {
|
|
|
376
378
|
}
|
|
377
379
|
}
|
|
378
380
|
// Parse workbook.xml to get sheet name -> rId mapping
|
|
379
|
-
const workbookXml = (0,
|
|
380
|
-
const sheets = (0,
|
|
381
|
+
const workbookXml = (0, xmlUtils_js_1.parseXmlString)(workbookFile.content.toString());
|
|
382
|
+
const sheets = (0, xmlUtils_js_1.getElementsByTagName)(workbookXml, "sheet");
|
|
381
383
|
for (const sheet of sheets) {
|
|
382
384
|
const name = sheet.getAttribute("name");
|
|
383
385
|
const rId = sheet.getAttribute("r:id");
|
|
@@ -420,10 +422,10 @@ const parseExcel = async (buffer, config) => {
|
|
|
420
422
|
if (cMatches) {
|
|
421
423
|
for (const cXml of cMatches) {
|
|
422
424
|
// Extract cell value
|
|
423
|
-
const typeMatch = cXml.match(/t="([a-
|
|
425
|
+
const typeMatch = cXml.match(/t="([a-zA-Z]+)"/);
|
|
424
426
|
const type = typeMatch ? typeMatch[1] : 'n'; // n = number (default)
|
|
425
|
-
const vMatch = cXml.match(/<v>(
|
|
426
|
-
const tMatch = cXml.match(/<t>(
|
|
427
|
+
const vMatch = cXml.match(/<v>([\s\S]*?)<\/v>/);
|
|
428
|
+
const tMatch = cXml.match(/<t>([\s\S]*?)<\/t>/);
|
|
427
429
|
let text = '';
|
|
428
430
|
let cellNodes = [];
|
|
429
431
|
if (type === 's' && vMatch) {
|
|
@@ -440,10 +442,10 @@ const parseExcel = async (buffer, config) => {
|
|
|
440
442
|
}
|
|
441
443
|
}
|
|
442
444
|
else if (type === 'inlineStr' && tMatch) {
|
|
443
|
-
text = tMatch[1];
|
|
445
|
+
text = tMatch[1].trim();
|
|
444
446
|
}
|
|
445
447
|
else if (vMatch) {
|
|
446
|
-
text = vMatch[1];
|
|
448
|
+
text = vMatch[1].trim();
|
|
447
449
|
}
|
|
448
450
|
// Parse cell coordinate
|
|
449
451
|
const coordMatch = cXml.match(/r="([A-Z]+)(\d+)"/);
|
|
@@ -515,8 +517,8 @@ const parseExcel = async (buffer, config) => {
|
|
|
515
517
|
const relsFile = files.find(f => f.path === relsFilename);
|
|
516
518
|
const drawingMap = {}; // rId -> drawingPath
|
|
517
519
|
if (relsFile) {
|
|
518
|
-
const relsXml = (0,
|
|
519
|
-
const relationships = (0,
|
|
520
|
+
const relsXml = (0, xmlUtils_js_1.parseXmlString)(relsFile.content.toString());
|
|
521
|
+
const relationships = (0, xmlUtils_js_1.getElementsByTagName)(relsXml, "Relationship");
|
|
520
522
|
for (const rel of relationships) {
|
|
521
523
|
const id = rel.getAttribute("Id");
|
|
522
524
|
const target = rel.getAttribute("Target");
|
|
@@ -588,7 +590,13 @@ const parseExcel = async (buffer, config) => {
|
|
|
588
590
|
}
|
|
589
591
|
}
|
|
590
592
|
const corePropsFile = files.find(f => f.path.match(corePropsFileRegex));
|
|
591
|
-
const metadata = corePropsFile ? (0,
|
|
593
|
+
const metadata = corePropsFile ? (0, xmlUtils_js_1.parseOfficeMetadata)(corePropsFile.content.toString()) : {};
|
|
594
|
+
const customPropsFile = files.find(f => f.path.match(customPropsFileRegex));
|
|
595
|
+
if (customPropsFile) {
|
|
596
|
+
const customProperties = (0, xmlUtils_js_1.parseOOXMLCustomProperties)(customPropsFile.content.toString());
|
|
597
|
+
if (Object.keys(customProperties).length > 0)
|
|
598
|
+
metadata.customProperties = customProperties;
|
|
599
|
+
}
|
|
592
600
|
// Link OCR text and chart data to content nodes (like PPTX parser)
|
|
593
601
|
const assignAttachmentData = (nodes) => {
|
|
594
602
|
for (const node of nodes) {
|
|
@@ -20,7 +20,7 @@
|
|
|
20
20
|
*
|
|
21
21
|
* @module OpenOfficeParser
|
|
22
22
|
*/
|
|
23
|
-
import { OfficeParserAST, OfficeParserConfig } from '../types';
|
|
23
|
+
import { OfficeParserAST, OfficeParserConfig } from '../types.js';
|
|
24
24
|
/**
|
|
25
25
|
* Parses an OpenOffice document (.odt, .odp, .ods) and extracts content.
|
|
26
26
|
*
|