officeparser 7.2.1 → 7.2.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +14 -8
- package/dist/defaults.js +7 -3
- package/dist/generators/ChunkingGenerator.js +1 -1
- package/dist/generators/HtmlGenerator.js +1 -1
- package/dist/officeparser.browser.d.ts +32 -2
- package/dist/officeparser.browser.iife.js +122 -122
- package/dist/officeparser.browser.mjs +122 -122
- package/dist/officeparser.browser.slim.d.ts +1992 -0
- package/dist/officeparser.browser.slim.iife.js +1182 -0
- package/dist/officeparser.browser.slim.mjs +1181 -0
- package/dist/parsers/ExcelParser.js +1 -1
- package/dist/parsers/OpenOfficeParser.js +242 -55
- package/dist/parsers/PowerPointParser.js +1 -1
- package/dist/parsers/WordParser.js +3 -3
- package/dist/types.d.ts +35 -2
- package/dist/types.js +8 -0
- package/dist/utils/configUtils.js +14 -2
- package/dist/utils/errorUtils.js +5 -1
- package/dist/utils/zipUtils.d.ts +2 -1
- package/dist/utils/zipUtils.js +34 -2
- package/package.json +12 -7
- package/dist/sbom.cdx.json +0 -1807
|
@@ -67,7 +67,7 @@ const parseExcel = async (buffer, config) => {
|
|
|
67
67
|
!!x.match(customPropsFileRegex) ||
|
|
68
68
|
!!x.match(appPropsFileRegex) ||
|
|
69
69
|
(!!config.extractAttachments && (!!x.match(mediaFileRegex) || !!x.match(drawingRelsRegex))) ||
|
|
70
|
-
((!!config.extractAttachments || !config.ignoreComments) && !!x.match(relsRegex)));
|
|
70
|
+
((!!config.extractAttachments || !config.ignoreComments) && !!x.match(relsRegex)), config.decompressionLimits);
|
|
71
71
|
const sharedStringsFile = files.find(f => f.path === stringsFilePath);
|
|
72
72
|
// Updated to store structured content (rich text runs) or simple string
|
|
73
73
|
const sharedStrings = [];
|
|
@@ -31,6 +31,16 @@ const imageUtils_js_1 = require("../utils/imageUtils.js");
|
|
|
31
31
|
const ocrUtils_js_1 = require("../utils/ocrUtils.js");
|
|
32
32
|
const xmlUtils_js_1 = require("../utils/xmlUtils.js");
|
|
33
33
|
const zipUtils_js_1 = require("../utils/zipUtils.js");
|
|
34
|
+
/**
|
|
35
|
+
* Helper to clean and extract attachment name from xlink:href or paths.
|
|
36
|
+
* Handles trailing slashes, leading "./", and subdirectories.
|
|
37
|
+
*/
|
|
38
|
+
const cleanAttachmentName = (href) => {
|
|
39
|
+
if (!href)
|
|
40
|
+
return '';
|
|
41
|
+
const cleaned = href.replace(/^\.\//, '').replace(/\/$/, '');
|
|
42
|
+
return cleaned.split('/').pop() || '';
|
|
43
|
+
};
|
|
34
44
|
/**
|
|
35
45
|
* Parses an OpenOffice document (.odt, .odp, .ods) and extracts content.
|
|
36
46
|
*
|
|
@@ -54,7 +64,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
54
64
|
!!x.match(metaFileRegex) ||
|
|
55
65
|
!!x.match(stylesFileRegex) ||
|
|
56
66
|
!!x.match(mimetypeFileRegex) ||
|
|
57
|
-
(!!config.extractAttachments && !!x.match(mediaFileRegex)));
|
|
67
|
+
(!!config.extractAttachments && !!x.match(mediaFileRegex)), config.decompressionLimits);
|
|
58
68
|
// 1. Determine File Type
|
|
59
69
|
const mimetypeFile = files.find(f => f.path === 'mimetype');
|
|
60
70
|
let fileType = 'odt'; // Default
|
|
@@ -82,6 +92,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
82
92
|
let lastListStyle = null;
|
|
83
93
|
let listIdCounter = 0;
|
|
84
94
|
let lastWasList = false;
|
|
95
|
+
let traverse;
|
|
85
96
|
// Helper to parse styles
|
|
86
97
|
const parseStyles = (xml) => {
|
|
87
98
|
const styles = (0, xmlUtils_js_1.getElementsByTagName)(xml, "style:style");
|
|
@@ -170,6 +181,64 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
170
181
|
* Returns the paragraph content without creating a content node.
|
|
171
182
|
*
|
|
172
183
|
* @param node - The paragraph element to parse
|
|
184
|
+
*/
|
|
185
|
+
const parseMathML = (node) => {
|
|
186
|
+
if (!node)
|
|
187
|
+
return '';
|
|
188
|
+
if (node.nodeType === 3) { // Text node
|
|
189
|
+
return node.textContent || '';
|
|
190
|
+
}
|
|
191
|
+
if (node.nodeType !== 1) { // Not an element
|
|
192
|
+
return '';
|
|
193
|
+
}
|
|
194
|
+
const element = node;
|
|
195
|
+
const tagName = element.tagName.toLowerCase().replace(/^.*:/, ''); // strip namespace prefix
|
|
196
|
+
switch (tagName) {
|
|
197
|
+
case 'math':
|
|
198
|
+
case 'mrow':
|
|
199
|
+
case 'semantics':
|
|
200
|
+
return Array.from(element.childNodes).map(parseMathML).join('');
|
|
201
|
+
case 'mfrac': {
|
|
202
|
+
const children = Array.from(element.childNodes).filter((n) => n.nodeType === 1);
|
|
203
|
+
if (children.length >= 2) {
|
|
204
|
+
return `(${parseMathML(children[0])})/(${parseMathML(children[1])})`;
|
|
205
|
+
}
|
|
206
|
+
return Array.from(element.childNodes).map(parseMathML).join('');
|
|
207
|
+
}
|
|
208
|
+
case 'msub': {
|
|
209
|
+
const children = Array.from(element.childNodes).filter((n) => n.nodeType === 1);
|
|
210
|
+
if (children.length >= 2) {
|
|
211
|
+
return `${parseMathML(children[0])}_${parseMathML(children[1])}`;
|
|
212
|
+
}
|
|
213
|
+
return Array.from(element.childNodes).map(parseMathML).join('');
|
|
214
|
+
}
|
|
215
|
+
case 'msup': {
|
|
216
|
+
const children = Array.from(element.childNodes).filter((n) => n.nodeType === 1);
|
|
217
|
+
if (children.length >= 2) {
|
|
218
|
+
return `${parseMathML(children[0])}^${parseMathML(children[1])}`;
|
|
219
|
+
}
|
|
220
|
+
return Array.from(element.childNodes).map(parseMathML).join('');
|
|
221
|
+
}
|
|
222
|
+
case 'msubsup': {
|
|
223
|
+
const children = Array.from(element.childNodes).filter((n) => n.nodeType === 1);
|
|
224
|
+
if (children.length >= 3) {
|
|
225
|
+
return `${parseMathML(children[0])}_${parseMathML(children[1])}^${parseMathML(children[2])}`;
|
|
226
|
+
}
|
|
227
|
+
return Array.from(element.childNodes).map(parseMathML).join('');
|
|
228
|
+
}
|
|
229
|
+
case 'mi':
|
|
230
|
+
case 'mn':
|
|
231
|
+
case 'mo':
|
|
232
|
+
case 'mtext':
|
|
233
|
+
case 'ms':
|
|
234
|
+
return element.textContent || '';
|
|
235
|
+
case 'annotation':
|
|
236
|
+
return '';
|
|
237
|
+
default:
|
|
238
|
+
return Array.from(element.childNodes).map(parseMathML).join('');
|
|
239
|
+
}
|
|
240
|
+
};
|
|
241
|
+
/**
|
|
173
242
|
* Helper to parse inline content (text, spans, links, notes, etc.) recursively.
|
|
174
243
|
*
|
|
175
244
|
* @param node - The element to parse (paragraph, span, or link)
|
|
@@ -325,41 +394,114 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
325
394
|
}
|
|
326
395
|
}
|
|
327
396
|
else if (tagName === 'draw:frame') {
|
|
328
|
-
// Inline image
|
|
329
397
|
const frame = element;
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
|
|
398
|
+
const drawTextBox = (0, xmlUtils_js_1.getFirstElementByTagName)(frame, "draw:text-box");
|
|
399
|
+
const drawObject = (0, xmlUtils_js_1.getFirstElementByTagName)(frame, "draw:object");
|
|
400
|
+
if (drawTextBox) {
|
|
401
|
+
const textBoxChildren = [];
|
|
402
|
+
traverse(drawTextBox, textBoxChildren, false, sourceXml);
|
|
403
|
+
children.push(...textBoxChildren);
|
|
404
|
+
const textBoxText = textBoxChildren.map(c => c.text || '').join('\n');
|
|
405
|
+
fullText += textBoxText;
|
|
336
406
|
}
|
|
337
|
-
else if (
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
407
|
+
else if (drawObject) {
|
|
408
|
+
const href = drawObject.getAttribute("xlink:href");
|
|
409
|
+
let isFormula = false;
|
|
410
|
+
let formulaText = '';
|
|
411
|
+
let attachmentName = '';
|
|
412
|
+
if (href) {
|
|
413
|
+
attachmentName = cleanAttachmentName(href);
|
|
414
|
+
const objectPath = `${attachmentName}/content.xml`;
|
|
415
|
+
const objectFile = files.find(f => f.path === objectPath || f.path.endsWith(objectPath));
|
|
416
|
+
if (objectFile) {
|
|
417
|
+
const objXml = (0, xmlUtils_js_1.parseXmlString)(objectFile.content.toString());
|
|
418
|
+
const mathNode = (0, xmlUtils_js_1.getFirstElementByTagName)(objXml, "math");
|
|
419
|
+
if (mathNode) {
|
|
420
|
+
isFormula = true;
|
|
421
|
+
formulaText = parseMathML(mathNode).trim();
|
|
422
|
+
}
|
|
423
|
+
}
|
|
424
|
+
}
|
|
425
|
+
if (isFormula) {
|
|
426
|
+
fullText += formulaText;
|
|
427
|
+
const textNode = {
|
|
428
|
+
type: 'text',
|
|
429
|
+
text: formulaText,
|
|
430
|
+
formatting: parentFormatting,
|
|
431
|
+
metadata: linkMetadata ? { ...linkMetadata } : undefined
|
|
432
|
+
};
|
|
433
|
+
if (config.includeRawContent) {
|
|
434
|
+
textNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frame, sourceXml, config);
|
|
435
|
+
}
|
|
436
|
+
children.push(textNode);
|
|
437
|
+
}
|
|
438
|
+
else {
|
|
439
|
+
// Standard inline image extraction fallback if object is not a formula
|
|
440
|
+
let altText = '';
|
|
441
|
+
const svgTitle = (0, xmlUtils_js_1.getFirstElementByTagName)(frame, "svg:title");
|
|
442
|
+
const svgDesc = (0, xmlUtils_js_1.getFirstElementByTagName)(frame, "svg:desc");
|
|
443
|
+
if (svgTitle && svgTitle.textContent) {
|
|
444
|
+
altText = svgTitle.textContent;
|
|
445
|
+
}
|
|
446
|
+
else if (svgDesc && svgDesc.textContent) {
|
|
447
|
+
altText = svgDesc.textContent;
|
|
448
|
+
}
|
|
449
|
+
let imageHref = '';
|
|
450
|
+
const drawImages = (0, xmlUtils_js_1.getElementsByTagName)(frame, "draw:image");
|
|
451
|
+
if (drawImages.length > 0) {
|
|
452
|
+
imageHref = drawImages[0].getAttribute("xlink:href") || '';
|
|
453
|
+
if (imageHref) {
|
|
454
|
+
imageHref = cleanAttachmentName(imageHref);
|
|
455
|
+
}
|
|
456
|
+
}
|
|
457
|
+
const imageNode = {
|
|
458
|
+
type: 'image',
|
|
459
|
+
text: '',
|
|
460
|
+
children: [],
|
|
461
|
+
metadata: {
|
|
462
|
+
attachmentName: imageHref || attachmentName,
|
|
463
|
+
...(altText ? { altText } : {})
|
|
464
|
+
}
|
|
465
|
+
};
|
|
466
|
+
if (config.includeRawContent) {
|
|
467
|
+
imageNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frame, sourceXml, config);
|
|
468
|
+
}
|
|
469
|
+
children.push(imageNode);
|
|
348
470
|
}
|
|
349
471
|
}
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
|
|
472
|
+
else {
|
|
473
|
+
// Standard inline image extraction fallback
|
|
474
|
+
let altText = '';
|
|
475
|
+
const svgTitle = (0, xmlUtils_js_1.getFirstElementByTagName)(frame, "svg:title");
|
|
476
|
+
const svgDesc = (0, xmlUtils_js_1.getFirstElementByTagName)(frame, "svg:desc");
|
|
477
|
+
if (svgTitle && svgTitle.textContent) {
|
|
478
|
+
altText = svgTitle.textContent;
|
|
357
479
|
}
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
|
|
480
|
+
else if (svgDesc && svgDesc.textContent) {
|
|
481
|
+
altText = svgDesc.textContent;
|
|
482
|
+
}
|
|
483
|
+
let imageHref = '';
|
|
484
|
+
const drawImages = (0, xmlUtils_js_1.getElementsByTagName)(frame, "draw:image");
|
|
485
|
+
if (drawImages.length > 0) {
|
|
486
|
+
imageHref = drawImages[0].getAttribute("xlink:href") || '';
|
|
487
|
+
if (imageHref) {
|
|
488
|
+
imageHref = cleanAttachmentName(imageHref);
|
|
489
|
+
}
|
|
490
|
+
}
|
|
491
|
+
const imageNode = {
|
|
492
|
+
type: 'image',
|
|
493
|
+
text: '',
|
|
494
|
+
children: [],
|
|
495
|
+
metadata: {
|
|
496
|
+
attachmentName: imageHref,
|
|
497
|
+
...(altText ? { altText } : {})
|
|
498
|
+
}
|
|
499
|
+
};
|
|
500
|
+
if (config.includeRawContent) {
|
|
501
|
+
imageNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frame, sourceXml, config);
|
|
502
|
+
}
|
|
503
|
+
children.push(imageNode);
|
|
361
504
|
}
|
|
362
|
-
children.push(imageNode);
|
|
363
505
|
}
|
|
364
506
|
}
|
|
365
507
|
}
|
|
@@ -713,7 +855,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
713
855
|
* @param sourceXml - The source XML string for raw content extraction
|
|
714
856
|
* @param asSheet - If true, treats tables as sheets (for ODS)
|
|
715
857
|
*/
|
|
716
|
-
|
|
858
|
+
traverse = (node, targetArray, forceHeading = false, sourceXml = '', asSheet = false) => {
|
|
717
859
|
if (node.tagName === "text:p") {
|
|
718
860
|
const pContent = parseParagraphContent(node, paragraphStyleMap, styleMap, config, sourceXml);
|
|
719
861
|
const type = (forceHeading || (node.getAttribute("text:style-name") || '').toLowerCase().includes('title')) ? 'heading' : 'paragraph';
|
|
@@ -1011,8 +1153,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1011
1153
|
// Extract image href to link to attachment
|
|
1012
1154
|
let imageHref = image.getAttribute("xlink:href") || '';
|
|
1013
1155
|
if (imageHref) {
|
|
1014
|
-
|
|
1015
|
-
imageHref = parts[parts.length - 1];
|
|
1156
|
+
imageHref = cleanAttachmentName(imageHref);
|
|
1016
1157
|
}
|
|
1017
1158
|
const metadata = {
|
|
1018
1159
|
attachmentName: imageHref,
|
|
@@ -1034,25 +1175,47 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1034
1175
|
targetArray.push(imageNode);
|
|
1035
1176
|
}
|
|
1036
1177
|
else if (object) {
|
|
1037
|
-
// Handle embedded objects like charts
|
|
1178
|
+
// Handle embedded objects like charts or math formulas
|
|
1038
1179
|
const href = object.getAttribute("xlink:href");
|
|
1039
1180
|
if (href) {
|
|
1040
|
-
const attachmentName = href
|
|
1181
|
+
const attachmentName = cleanAttachmentName(href);
|
|
1041
1182
|
const objectPath = `${attachmentName}/content.xml`;
|
|
1042
1183
|
const objectFile = files.find(f => f.path === objectPath || f.path.endsWith(objectPath));
|
|
1043
1184
|
if (objectFile) {
|
|
1044
|
-
const
|
|
1045
|
-
const
|
|
1046
|
-
|
|
1047
|
-
|
|
1048
|
-
|
|
1049
|
-
|
|
1050
|
-
|
|
1185
|
+
const objXml = (0, xmlUtils_js_1.parseXmlString)(objectFile.content.toString());
|
|
1186
|
+
const mathNode = (0, xmlUtils_js_1.getFirstElementByTagName)(objXml, "math");
|
|
1187
|
+
if (mathNode) {
|
|
1188
|
+
// Math formula object at block level
|
|
1189
|
+
const formulaText = parseMathML(mathNode).trim();
|
|
1190
|
+
const formulaNode = {
|
|
1191
|
+
type: 'paragraph',
|
|
1192
|
+
text: formulaText,
|
|
1193
|
+
children: [
|
|
1194
|
+
{
|
|
1195
|
+
type: 'text',
|
|
1196
|
+
text: formulaText
|
|
1197
|
+
}
|
|
1198
|
+
]
|
|
1199
|
+
};
|
|
1200
|
+
if (config.includeRawContent) {
|
|
1201
|
+
formulaNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
|
|
1051
1202
|
}
|
|
1052
|
-
|
|
1053
|
-
|
|
1054
|
-
|
|
1055
|
-
|
|
1203
|
+
targetArray.push(formulaNode);
|
|
1204
|
+
}
|
|
1205
|
+
else {
|
|
1206
|
+
const chartData = (0, chartUtils_js_1.extractChartData)(objectFile.content);
|
|
1207
|
+
const chartNode = {
|
|
1208
|
+
type: 'chart',
|
|
1209
|
+
text: chartData.rawTexts.join(" "),
|
|
1210
|
+
metadata: {
|
|
1211
|
+
attachmentName: attachmentName,
|
|
1212
|
+
chartData
|
|
1213
|
+
}
|
|
1214
|
+
};
|
|
1215
|
+
if (config.includeRawContent)
|
|
1216
|
+
chartNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
|
|
1217
|
+
targetArray.push(chartNode);
|
|
1218
|
+
}
|
|
1056
1219
|
}
|
|
1057
1220
|
else {
|
|
1058
1221
|
const chartNode = {
|
|
@@ -1077,8 +1240,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1077
1240
|
}
|
|
1078
1241
|
}
|
|
1079
1242
|
}
|
|
1080
|
-
}
|
|
1081
|
-
;
|
|
1243
|
+
};
|
|
1082
1244
|
// ODS: Spreadsheet
|
|
1083
1245
|
if (fileType === 'ods') {
|
|
1084
1246
|
const spreadsheet = (0, xmlUtils_js_1.getFirstElementByTagName)(body, "office:spreadsheet");
|
|
@@ -1156,28 +1318,50 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1156
1318
|
if (drawImages.length > 0) {
|
|
1157
1319
|
const rawHref = drawImages[0].getAttribute("xlink:href");
|
|
1158
1320
|
if (rawHref) {
|
|
1159
|
-
|
|
1160
|
-
imageHref = parts[parts.length - 1];
|
|
1321
|
+
imageHref = cleanAttachmentName(rawHref);
|
|
1161
1322
|
}
|
|
1162
1323
|
}
|
|
1163
|
-
// Extract chart object href
|
|
1324
|
+
// Extract chart or math object href
|
|
1164
1325
|
let chartHref = '';
|
|
1326
|
+
let isFormula = false;
|
|
1327
|
+
let formulaText = '';
|
|
1165
1328
|
const drawObjects = (0, xmlUtils_js_1.getElementsByTagName)(frame, "draw:object");
|
|
1166
1329
|
if (drawObjects.length > 0) {
|
|
1167
1330
|
const href = drawObjects[0].getAttribute("xlink:href");
|
|
1168
1331
|
if (href) {
|
|
1169
|
-
|
|
1170
|
-
|
|
1332
|
+
chartHref = cleanAttachmentName(href);
|
|
1333
|
+
const objectPath = `${chartHref}/content.xml`;
|
|
1334
|
+
const objectFile = files.find(f => f.path === objectPath || f.path.endsWith(objectPath));
|
|
1335
|
+
if (objectFile) {
|
|
1336
|
+
const objXml = (0, xmlUtils_js_1.parseXmlString)(objectFile.content.toString());
|
|
1337
|
+
const mathNode = (0, xmlUtils_js_1.getFirstElementByTagName)(objXml, "math");
|
|
1338
|
+
if (mathNode) {
|
|
1339
|
+
isFormula = true;
|
|
1340
|
+
formulaText = parseMathML(mathNode).trim();
|
|
1341
|
+
}
|
|
1342
|
+
}
|
|
1171
1343
|
}
|
|
1172
1344
|
}
|
|
1173
|
-
if (
|
|
1345
|
+
if (isFormula) {
|
|
1346
|
+
cellText += formulaText;
|
|
1347
|
+
const textNode = {
|
|
1348
|
+
type: 'text',
|
|
1349
|
+
text: formulaText,
|
|
1350
|
+
formatting: {}
|
|
1351
|
+
};
|
|
1352
|
+
if (config.includeRawContent) {
|
|
1353
|
+
textNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frame, xmlString, config);
|
|
1354
|
+
}
|
|
1355
|
+
children.push(textNode);
|
|
1356
|
+
}
|
|
1357
|
+
else if (drawImages.length > 0) {
|
|
1174
1358
|
// logic for image node
|
|
1175
1359
|
const imageNode = {
|
|
1176
1360
|
type: 'image',
|
|
1177
1361
|
text: '', // Will be populated by assignAttachmentData
|
|
1178
1362
|
children: [],
|
|
1179
1363
|
metadata: {
|
|
1180
|
-
attachmentName: imageHref, // Might be empty, will resolve in assignAttachmentData
|
|
1364
|
+
attachmentName: imageHref || chartHref, // Might be empty, will resolve in assignAttachmentData
|
|
1181
1365
|
...(altText ? { altText } : {})
|
|
1182
1366
|
}
|
|
1183
1367
|
};
|
|
@@ -1195,6 +1379,9 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1195
1379
|
attachmentName: chartHref
|
|
1196
1380
|
}
|
|
1197
1381
|
};
|
|
1382
|
+
if (config.includeRawContent) {
|
|
1383
|
+
chartNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frame, xmlString, config);
|
|
1384
|
+
}
|
|
1198
1385
|
children.push(chartNode);
|
|
1199
1386
|
}
|
|
1200
1387
|
}
|
|
@@ -63,7 +63,7 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
63
63
|
!!x.match(slideRelsRegex) ||
|
|
64
64
|
(!config.ignoreComments && (!!x.match(commentsFileRegex) || !!x.match(commentAuthorsRegex))) ||
|
|
65
65
|
(!config.ignoreSlideMasters && !!x.match(slideMastersRegex)) ||
|
|
66
|
-
(!!config.extractAttachments && (!!x.match(mediaFileRegex) || !!x.match(chartFileRegex))));
|
|
66
|
+
(!!config.extractAttachments && (!!x.match(mediaFileRegex) || !!x.match(chartFileRegex))), config.decompressionLimits);
|
|
67
67
|
// Extract metadata
|
|
68
68
|
const corePropsFile = files.find(f => f.path.match(corePropsFileRegex));
|
|
69
69
|
const metadata = corePropsFile ? (0, xmlUtils_js_1.parseOfficeMetadata)(corePropsFile.content.toString()) : {};
|
|
@@ -243,7 +243,7 @@ const parseWord = async (buffer, config) => {
|
|
|
243
243
|
!!x.match(appPropsFileRegex) ||
|
|
244
244
|
!!x.match(relsFileRegex) ||
|
|
245
245
|
!!x.match(stylesFileRegex) ||
|
|
246
|
-
(!!config.extractAttachments && !!x.match(mediaFileRegex)));
|
|
246
|
+
(!!config.extractAttachments && !!x.match(mediaFileRegex)), config.decompressionLimits);
|
|
247
247
|
// Extract metadata
|
|
248
248
|
const corePropsFile = files.find(f => f.path.match(corePropsFileRegex));
|
|
249
249
|
const metadata = corePropsFile ? (0, xmlUtils_js_1.parseOfficeMetadata)(corePropsFile.content.toString()) : {};
|
|
@@ -473,7 +473,7 @@ const parseWord = async (buffer, config) => {
|
|
|
473
473
|
const comments = [];
|
|
474
474
|
// Traverse children of paragraph (runs, hyperlinks, etc.)
|
|
475
475
|
const processChildNode = (node) => {
|
|
476
|
-
if ((0, xmlUtils_js_1.isElement)(node) && node.nodeName === 'w:r') {
|
|
476
|
+
if ((0, xmlUtils_js_1.isElement)(node) && (node.nodeName === 'w:r' || node.nodeName === 'm:r')) {
|
|
477
477
|
const runNode = node;
|
|
478
478
|
const rPr = (0, xmlUtils_js_1.getFirstElementByTagName)(runNode, "w:rPr");
|
|
479
479
|
// Formatting
|
|
@@ -512,7 +512,7 @@ const parseWord = async (buffer, config) => {
|
|
|
512
512
|
continue;
|
|
513
513
|
// also handle unprefixed version (mirroring the behaviour of getElementsByTagName)
|
|
514
514
|
// Text content
|
|
515
|
-
if (child.tagName === "w:t" || child.tagName === "t") {
|
|
515
|
+
if (child.tagName === "w:t" || child.tagName === "t" || child.tagName === "m:t") {
|
|
516
516
|
const tNode = child;
|
|
517
517
|
const tContent = tNode.textContent || '';
|
|
518
518
|
text += tContent;
|
package/dist/types.d.ts
CHANGED
|
@@ -32,7 +32,15 @@ export declare enum OfficeErrorType {
|
|
|
32
32
|
/** Semantic chunking strategy is selected but no embedding function is provided */
|
|
33
33
|
MISSING_EMBEDDING_FUNCTION = "MISSING_EMBEDDING_FUNCTION",
|
|
34
34
|
/** The operation was aborted */
|
|
35
|
-
OPERATION_ABORTED = "OPERATION_ABORTED"
|
|
35
|
+
OPERATION_ABORTED = "OPERATION_ABORTED",
|
|
36
|
+
/** ZIP entry count exceeds limit */
|
|
37
|
+
ZIP_ENTRY_COUNT_LIMIT_EXCEEDED = "ZIP_ENTRY_COUNT_LIMIT_EXCEEDED",
|
|
38
|
+
/** ZIP entry missing a valid declared size */
|
|
39
|
+
ZIP_ENTRY_INVALID_SIZE = "ZIP_ENTRY_INVALID_SIZE",
|
|
40
|
+
/** ZIP uncompressed size limit exceeded */
|
|
41
|
+
ZIP_SIZE_LIMIT_EXCEEDED = "ZIP_SIZE_LIMIT_EXCEEDED",
|
|
42
|
+
/** Embedding call timed out */
|
|
43
|
+
EMBEDDING_TIMEOUT = "EMBEDDING_TIMEOUT"
|
|
36
44
|
}
|
|
37
45
|
/**
|
|
38
46
|
* Standard warning types for OfficeParser.
|
|
@@ -309,7 +317,7 @@ export interface OfficeParserConfig {
|
|
|
309
317
|
* The URL/path to the PDF.js worker script.
|
|
310
318
|
*
|
|
311
319
|
* **Mandatory** when using PDF parsing in browser environments to avoid worker configuration errors.
|
|
312
|
-
* If not provided, it defaults to `https://cdn.jsdelivr.net/npm/pdfjs-dist@
|
|
320
|
+
* If not provided, it defaults to `https://cdn.jsdelivr.net/npm/pdfjs-dist@6.1.200/build/pdf.worker.min.mjs`.
|
|
313
321
|
* You can override this with your own local path or a different CDN link.
|
|
314
322
|
*/
|
|
315
323
|
pdfWorkerSrc?: string;
|
|
@@ -347,6 +355,28 @@ export interface OfficeParserConfig {
|
|
|
347
355
|
* Defaults to ',' but can be overridden (e.g., ';', '\t').
|
|
348
356
|
*/
|
|
349
357
|
csvDelimiter?: string;
|
|
358
|
+
/**
|
|
359
|
+
* Limits and checks applied during ZIP extraction to protect against excessive
|
|
360
|
+
* memory and resource usage.
|
|
361
|
+
*/
|
|
362
|
+
decompressionLimits?: DecompressionLimits;
|
|
363
|
+
}
|
|
364
|
+
/**
|
|
365
|
+
* Limits applied to ZIP archive decompression.
|
|
366
|
+
*/
|
|
367
|
+
export interface DecompressionLimits {
|
|
368
|
+
/**
|
|
369
|
+
* Maximum allowed total uncompressed size (in bytes) of files extracted from a ZIP archive.
|
|
370
|
+
* Applies to OOXML (DOCX, XLSX, PPTX) and ODF (ODT, ODP, ODS) formats.
|
|
371
|
+
* Default is 536870912 (512 MB).
|
|
372
|
+
*/
|
|
373
|
+
maxUncompressedBytes?: number;
|
|
374
|
+
/**
|
|
375
|
+
* Maximum allowed number of entries (files and directories) in a ZIP archive.
|
|
376
|
+
* Applies to OOXML (DOCX, XLSX, PPTX) and ODF (ODT, ODP, ODS) formats.
|
|
377
|
+
* Default is 10000.
|
|
378
|
+
*/
|
|
379
|
+
maxZipEntries?: number;
|
|
350
380
|
}
|
|
351
381
|
/**
|
|
352
382
|
* A fully-populated parser configuration containing all options.
|
|
@@ -1842,4 +1872,7 @@ export interface OfficeParserAST {
|
|
|
1842
1872
|
*/
|
|
1843
1873
|
to<T extends this, D extends SupportedDestination<T['type']>>(this: T, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult<D>>;
|
|
1844
1874
|
}
|
|
1875
|
+
declare global {
|
|
1876
|
+
const __SLIM__: boolean | undefined;
|
|
1877
|
+
}
|
|
1845
1878
|
export {};
|
package/dist/types.js
CHANGED
|
@@ -37,6 +37,14 @@ var OfficeErrorType;
|
|
|
37
37
|
OfficeErrorType["MISSING_EMBEDDING_FUNCTION"] = "MISSING_EMBEDDING_FUNCTION";
|
|
38
38
|
/** The operation was aborted */
|
|
39
39
|
OfficeErrorType["OPERATION_ABORTED"] = "OPERATION_ABORTED";
|
|
40
|
+
/** ZIP entry count exceeds limit */
|
|
41
|
+
OfficeErrorType["ZIP_ENTRY_COUNT_LIMIT_EXCEEDED"] = "ZIP_ENTRY_COUNT_LIMIT_EXCEEDED";
|
|
42
|
+
/** ZIP entry missing a valid declared size */
|
|
43
|
+
OfficeErrorType["ZIP_ENTRY_INVALID_SIZE"] = "ZIP_ENTRY_INVALID_SIZE";
|
|
44
|
+
/** ZIP uncompressed size limit exceeded */
|
|
45
|
+
OfficeErrorType["ZIP_SIZE_LIMIT_EXCEEDED"] = "ZIP_SIZE_LIMIT_EXCEEDED";
|
|
46
|
+
/** Embedding call timed out */
|
|
47
|
+
OfficeErrorType["EMBEDDING_TIMEOUT"] = "EMBEDDING_TIMEOUT";
|
|
40
48
|
})(OfficeErrorType || (exports.OfficeErrorType = OfficeErrorType = {}));
|
|
41
49
|
/**
|
|
42
50
|
* Standard warning types for OfficeParser.
|
|
@@ -57,6 +57,12 @@ function isFullParserConfig(config) {
|
|
|
57
57
|
*/
|
|
58
58
|
function resolveParserConfig(userConfig) {
|
|
59
59
|
if (isFullParserConfig(userConfig)) {
|
|
60
|
+
if (!userConfig.decompressionLimits) {
|
|
61
|
+
userConfig.decompressionLimits = {
|
|
62
|
+
maxUncompressedBytes: 512 * 1024 * 1024,
|
|
63
|
+
maxZipEntries: 10000,
|
|
64
|
+
};
|
|
65
|
+
}
|
|
60
66
|
return userConfig;
|
|
61
67
|
}
|
|
62
68
|
// 1. Start with full defaults (deep cloned)
|
|
@@ -65,9 +71,15 @@ function resolveParserConfig(userConfig) {
|
|
|
65
71
|
return config;
|
|
66
72
|
}
|
|
67
73
|
// 2. Merge user config
|
|
68
|
-
// We handle ocrConfig specially to avoid shallow-overwriting the whole
|
|
69
|
-
const { ocrConfig, ...rest } = userConfig;
|
|
74
|
+
// We handle ocrConfig and decompressionLimits specially to avoid shallow-overwriting the whole objects
|
|
75
|
+
const { ocrConfig, decompressionLimits, ...rest } = userConfig;
|
|
70
76
|
Object.assign(config, rest);
|
|
77
|
+
if (decompressionLimits) {
|
|
78
|
+
config.decompressionLimits = {
|
|
79
|
+
...config.decompressionLimits,
|
|
80
|
+
...decompressionLimits,
|
|
81
|
+
};
|
|
82
|
+
}
|
|
71
83
|
if (ocrConfig) {
|
|
72
84
|
const { timeout, ...ocrRest } = ocrConfig;
|
|
73
85
|
config.ocrConfig = {
|
package/dist/utils/errorUtils.js
CHANGED
|
@@ -30,7 +30,11 @@ const ERROR_MESSAGES = {
|
|
|
30
30
|
[types_js_1.OfficeErrorType.INVALID_SELECTOR]: (selector) => `Invalid selector: ${selector}`,
|
|
31
31
|
[types_js_1.OfficeErrorType.INVALID_OUTPUT_MAPPING]: (output) => `Invalid output mapping: ${output}`,
|
|
32
32
|
[types_js_1.OfficeErrorType.MISSING_EMBEDDING_FUNCTION]: `Semantic chunking requires an "embeddingFunction" to be provided in chunksConfig. This function must accept a string and return a Promise resolving to a number array (vector).`,
|
|
33
|
-
[types_js_1.OfficeErrorType.OPERATION_ABORTED]: `The operation was aborted
|
|
33
|
+
[types_js_1.OfficeErrorType.OPERATION_ABORTED]: `The operation was aborted.`,
|
|
34
|
+
[types_js_1.OfficeErrorType.ZIP_ENTRY_COUNT_LIMIT_EXCEEDED]: (limit) => `ZIP entry count exceeds limit (${limit})`,
|
|
35
|
+
[types_js_1.OfficeErrorType.ZIP_ENTRY_INVALID_SIZE]: `ZIP entry missing a valid declared size`,
|
|
36
|
+
[types_js_1.OfficeErrorType.ZIP_SIZE_LIMIT_EXCEEDED]: (limit) => `ZIP uncompressed size limit exceeded (${limit} bytes)`,
|
|
37
|
+
[types_js_1.OfficeErrorType.EMBEDDING_TIMEOUT]: (timeout) => `Embedding call timed out after ${timeout}ms`
|
|
34
38
|
};
|
|
35
39
|
/**
|
|
36
40
|
* Lookup table for warning messages.
|
package/dist/utils/zipUtils.d.ts
CHANGED
|
@@ -13,6 +13,7 @@
|
|
|
13
13
|
*
|
|
14
14
|
* @module zipUtils
|
|
15
15
|
*/
|
|
16
|
+
import { DecompressionLimits } from '../types.js';
|
|
16
17
|
/**
|
|
17
18
|
* Represents a file extracted from a ZIP archive.
|
|
18
19
|
* Contains the file's path within the archive and its content as a Buffer.
|
|
@@ -69,5 +70,5 @@ interface ZipFileContent {
|
|
|
69
70
|
*
|
|
70
71
|
* @see https://pkware.cachefly.net/webdocs/casestudies/APPNOTE.TXT ZIP file format specification
|
|
71
72
|
*/
|
|
72
|
-
export declare const extractFiles: (zipInput: Buffer, filterFn: (fileName: string) => boolean) => Promise<ZipFileContent[]>;
|
|
73
|
+
export declare const extractFiles: (zipInput: Buffer, filterFn: (fileName: string) => boolean, limits: DecompressionLimits) => Promise<ZipFileContent[]>;
|
|
73
74
|
export {};
|
package/dist/utils/zipUtils.js
CHANGED
|
@@ -17,6 +17,8 @@
|
|
|
17
17
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
18
18
|
exports.extractFiles = void 0;
|
|
19
19
|
const fflate_1 = require("fflate");
|
|
20
|
+
const types_js_1 = require("../types.js");
|
|
21
|
+
const errorUtils_js_1 = require("./errorUtils.js");
|
|
20
22
|
/**
|
|
21
23
|
* Extracts files from a ZIP archive with optional filtering.
|
|
22
24
|
*
|
|
@@ -56,9 +58,39 @@ const fflate_1 = require("fflate");
|
|
|
56
58
|
*
|
|
57
59
|
* @see https://pkware.cachefly.net/webdocs/casestudies/APPNOTE.TXT ZIP file format specification
|
|
58
60
|
*/
|
|
59
|
-
const extractFiles = (zipInput, filterFn) => {
|
|
61
|
+
const extractFiles = (zipInput, filterFn, limits) => {
|
|
62
|
+
const maxUncompressedBytes = limits?.maxUncompressedBytes !== undefined && Number.isFinite(limits.maxUncompressedBytes) && limits.maxUncompressedBytes >= 0
|
|
63
|
+
? limits.maxUncompressedBytes
|
|
64
|
+
: 512 * 1024 * 1024;
|
|
65
|
+
const maxZipEntries = limits?.maxZipEntries !== undefined && Number.isFinite(limits.maxZipEntries) && limits.maxZipEntries >= 0
|
|
66
|
+
? limits.maxZipEntries
|
|
67
|
+
: 10000;
|
|
60
68
|
return new Promise((resolve, reject) => {
|
|
61
|
-
|
|
69
|
+
let totalEntryCount = 0;
|
|
70
|
+
let entryCount = 0;
|
|
71
|
+
let declaredTotal = 0;
|
|
72
|
+
(0, fflate_1.unzip)(new Uint8Array(zipInput.buffer, zipInput.byteOffset, zipInput.byteLength), {
|
|
73
|
+
filter: (file) => {
|
|
74
|
+
totalEntryCount++;
|
|
75
|
+
if (totalEntryCount > maxZipEntries) {
|
|
76
|
+
reject((0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.ZIP_ENTRY_COUNT_LIMIT_EXCEEDED, undefined, maxZipEntries));
|
|
77
|
+
return false;
|
|
78
|
+
}
|
|
79
|
+
if (!filterFn(file.name))
|
|
80
|
+
return false;
|
|
81
|
+
if (typeof file.originalSize !== 'number' || !Number.isFinite(file.originalSize) || file.originalSize < 0) {
|
|
82
|
+
reject((0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.ZIP_ENTRY_INVALID_SIZE));
|
|
83
|
+
return false;
|
|
84
|
+
}
|
|
85
|
+
entryCount++;
|
|
86
|
+
declaredTotal += file.originalSize;
|
|
87
|
+
if (declaredTotal > maxUncompressedBytes) {
|
|
88
|
+
reject((0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.ZIP_SIZE_LIMIT_EXCEEDED, undefined, maxUncompressedBytes));
|
|
89
|
+
return false;
|
|
90
|
+
}
|
|
91
|
+
return true;
|
|
92
|
+
}
|
|
93
|
+
}, (err, decompressed) => {
|
|
62
94
|
if (err)
|
|
63
95
|
return reject(err);
|
|
64
96
|
resolve(Object.entries(decompressed).map(([path, data]) => ({
|