officeparser 7.2.1 → 7.2.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -67,7 +67,7 @@ const parseExcel = async (buffer, config) => {
67
67
  !!x.match(customPropsFileRegex) ||
68
68
  !!x.match(appPropsFileRegex) ||
69
69
  (!!config.extractAttachments && (!!x.match(mediaFileRegex) || !!x.match(drawingRelsRegex))) ||
70
- ((!!config.extractAttachments || !config.ignoreComments) && !!x.match(relsRegex)));
70
+ ((!!config.extractAttachments || !config.ignoreComments) && !!x.match(relsRegex)), config.decompressionLimits);
71
71
  const sharedStringsFile = files.find(f => f.path === stringsFilePath);
72
72
  // Updated to store structured content (rich text runs) or simple string
73
73
  const sharedStrings = [];
@@ -31,6 +31,16 @@ const imageUtils_js_1 = require("../utils/imageUtils.js");
31
31
  const ocrUtils_js_1 = require("../utils/ocrUtils.js");
32
32
  const xmlUtils_js_1 = require("../utils/xmlUtils.js");
33
33
  const zipUtils_js_1 = require("../utils/zipUtils.js");
34
+ /**
35
+ * Helper to clean and extract attachment name from xlink:href or paths.
36
+ * Handles trailing slashes, leading "./", and subdirectories.
37
+ */
38
+ const cleanAttachmentName = (href) => {
39
+ if (!href)
40
+ return '';
41
+ const cleaned = href.replace(/^\.\//, '').replace(/\/$/, '');
42
+ return cleaned.split('/').pop() || '';
43
+ };
34
44
  /**
35
45
  * Parses an OpenOffice document (.odt, .odp, .ods) and extracts content.
36
46
  *
@@ -54,7 +64,7 @@ const parseOpenOffice = async (buffer, config) => {
54
64
  !!x.match(metaFileRegex) ||
55
65
  !!x.match(stylesFileRegex) ||
56
66
  !!x.match(mimetypeFileRegex) ||
57
- (!!config.extractAttachments && !!x.match(mediaFileRegex)));
67
+ (!!config.extractAttachments && !!x.match(mediaFileRegex)), config.decompressionLimits);
58
68
  // 1. Determine File Type
59
69
  const mimetypeFile = files.find(f => f.path === 'mimetype');
60
70
  let fileType = 'odt'; // Default
@@ -82,6 +92,7 @@ const parseOpenOffice = async (buffer, config) => {
82
92
  let lastListStyle = null;
83
93
  let listIdCounter = 0;
84
94
  let lastWasList = false;
95
+ let traverse;
85
96
  // Helper to parse styles
86
97
  const parseStyles = (xml) => {
87
98
  const styles = (0, xmlUtils_js_1.getElementsByTagName)(xml, "style:style");
@@ -170,6 +181,64 @@ const parseOpenOffice = async (buffer, config) => {
170
181
  * Returns the paragraph content without creating a content node.
171
182
  *
172
183
  * @param node - The paragraph element to parse
184
+ */
185
+ const parseMathML = (node) => {
186
+ if (!node)
187
+ return '';
188
+ if (node.nodeType === 3) { // Text node
189
+ return node.textContent || '';
190
+ }
191
+ if (node.nodeType !== 1) { // Not an element
192
+ return '';
193
+ }
194
+ const element = node;
195
+ const tagName = element.tagName.toLowerCase().replace(/^.*:/, ''); // strip namespace prefix
196
+ switch (tagName) {
197
+ case 'math':
198
+ case 'mrow':
199
+ case 'semantics':
200
+ return Array.from(element.childNodes).map(parseMathML).join('');
201
+ case 'mfrac': {
202
+ const children = Array.from(element.childNodes).filter((n) => n.nodeType === 1);
203
+ if (children.length >= 2) {
204
+ return `(${parseMathML(children[0])})/(${parseMathML(children[1])})`;
205
+ }
206
+ return Array.from(element.childNodes).map(parseMathML).join('');
207
+ }
208
+ case 'msub': {
209
+ const children = Array.from(element.childNodes).filter((n) => n.nodeType === 1);
210
+ if (children.length >= 2) {
211
+ return `${parseMathML(children[0])}_${parseMathML(children[1])}`;
212
+ }
213
+ return Array.from(element.childNodes).map(parseMathML).join('');
214
+ }
215
+ case 'msup': {
216
+ const children = Array.from(element.childNodes).filter((n) => n.nodeType === 1);
217
+ if (children.length >= 2) {
218
+ return `${parseMathML(children[0])}^${parseMathML(children[1])}`;
219
+ }
220
+ return Array.from(element.childNodes).map(parseMathML).join('');
221
+ }
222
+ case 'msubsup': {
223
+ const children = Array.from(element.childNodes).filter((n) => n.nodeType === 1);
224
+ if (children.length >= 3) {
225
+ return `${parseMathML(children[0])}_${parseMathML(children[1])}^${parseMathML(children[2])}`;
226
+ }
227
+ return Array.from(element.childNodes).map(parseMathML).join('');
228
+ }
229
+ case 'mi':
230
+ case 'mn':
231
+ case 'mo':
232
+ case 'mtext':
233
+ case 'ms':
234
+ return element.textContent || '';
235
+ case 'annotation':
236
+ return '';
237
+ default:
238
+ return Array.from(element.childNodes).map(parseMathML).join('');
239
+ }
240
+ };
241
+ /**
173
242
  * Helper to parse inline content (text, spans, links, notes, etc.) recursively.
174
243
  *
175
244
  * @param node - The element to parse (paragraph, span, or link)
@@ -325,41 +394,114 @@ const parseOpenOffice = async (buffer, config) => {
325
394
  }
326
395
  }
327
396
  else if (tagName === 'draw:frame') {
328
- // Inline image
329
397
  const frame = element;
330
- // Extract alt text
331
- let altText = '';
332
- const svgTitle = (0, xmlUtils_js_1.getFirstElementByTagName)(frame, "svg:title");
333
- const svgDesc = (0, xmlUtils_js_1.getFirstElementByTagName)(frame, "svg:desc");
334
- if (svgTitle && svgTitle.textContent) {
335
- altText = svgTitle.textContent;
398
+ const drawTextBox = (0, xmlUtils_js_1.getFirstElementByTagName)(frame, "draw:text-box");
399
+ const drawObject = (0, xmlUtils_js_1.getFirstElementByTagName)(frame, "draw:object");
400
+ if (drawTextBox) {
401
+ const textBoxChildren = [];
402
+ traverse(drawTextBox, textBoxChildren, false, sourceXml);
403
+ children.push(...textBoxChildren);
404
+ const textBoxText = textBoxChildren.map(c => c.text || '').join('\n');
405
+ fullText += textBoxText;
336
406
  }
337
- else if (svgDesc && svgDesc.textContent) {
338
- altText = svgDesc.textContent;
339
- }
340
- // Extract image href
341
- let imageHref = '';
342
- const drawImages = (0, xmlUtils_js_1.getElementsByTagName)(frame, "draw:image");
343
- if (drawImages.length > 0) {
344
- imageHref = drawImages[0].getAttribute("xlink:href") || '';
345
- if (imageHref) {
346
- const parts = imageHref.split('/');
347
- imageHref = parts[parts.length - 1];
407
+ else if (drawObject) {
408
+ const href = drawObject.getAttribute("xlink:href");
409
+ let isFormula = false;
410
+ let formulaText = '';
411
+ let attachmentName = '';
412
+ if (href) {
413
+ attachmentName = cleanAttachmentName(href);
414
+ const objectPath = `${attachmentName}/content.xml`;
415
+ const objectFile = files.find(f => f.path === objectPath || f.path.endsWith(objectPath));
416
+ if (objectFile) {
417
+ const objXml = (0, xmlUtils_js_1.parseXmlString)(objectFile.content.toString());
418
+ const mathNode = (0, xmlUtils_js_1.getFirstElementByTagName)(objXml, "math");
419
+ if (mathNode) {
420
+ isFormula = true;
421
+ formulaText = parseMathML(mathNode).trim();
422
+ }
423
+ }
424
+ }
425
+ if (isFormula) {
426
+ fullText += formulaText;
427
+ const textNode = {
428
+ type: 'text',
429
+ text: formulaText,
430
+ formatting: parentFormatting,
431
+ metadata: linkMetadata ? { ...linkMetadata } : undefined
432
+ };
433
+ if (config.includeRawContent) {
434
+ textNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frame, sourceXml, config);
435
+ }
436
+ children.push(textNode);
437
+ }
438
+ else {
439
+ // Standard inline image extraction fallback if object is not a formula
440
+ let altText = '';
441
+ const svgTitle = (0, xmlUtils_js_1.getFirstElementByTagName)(frame, "svg:title");
442
+ const svgDesc = (0, xmlUtils_js_1.getFirstElementByTagName)(frame, "svg:desc");
443
+ if (svgTitle && svgTitle.textContent) {
444
+ altText = svgTitle.textContent;
445
+ }
446
+ else if (svgDesc && svgDesc.textContent) {
447
+ altText = svgDesc.textContent;
448
+ }
449
+ let imageHref = '';
450
+ const drawImages = (0, xmlUtils_js_1.getElementsByTagName)(frame, "draw:image");
451
+ if (drawImages.length > 0) {
452
+ imageHref = drawImages[0].getAttribute("xlink:href") || '';
453
+ if (imageHref) {
454
+ imageHref = cleanAttachmentName(imageHref);
455
+ }
456
+ }
457
+ const imageNode = {
458
+ type: 'image',
459
+ text: '',
460
+ children: [],
461
+ metadata: {
462
+ attachmentName: imageHref || attachmentName,
463
+ ...(altText ? { altText } : {})
464
+ }
465
+ };
466
+ if (config.includeRawContent) {
467
+ imageNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frame, sourceXml, config);
468
+ }
469
+ children.push(imageNode);
348
470
  }
349
471
  }
350
- const imageNode = {
351
- type: 'image',
352
- text: '',
353
- children: [],
354
- metadata: {
355
- attachmentName: imageHref,
356
- ...(altText ? { altText } : {})
472
+ else {
473
+ // Standard inline image extraction fallback
474
+ let altText = '';
475
+ const svgTitle = (0, xmlUtils_js_1.getFirstElementByTagName)(frame, "svg:title");
476
+ const svgDesc = (0, xmlUtils_js_1.getFirstElementByTagName)(frame, "svg:desc");
477
+ if (svgTitle && svgTitle.textContent) {
478
+ altText = svgTitle.textContent;
357
479
  }
358
- };
359
- if (config.includeRawContent) {
360
- imageNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frame, sourceXml, config);
480
+ else if (svgDesc && svgDesc.textContent) {
481
+ altText = svgDesc.textContent;
482
+ }
483
+ let imageHref = '';
484
+ const drawImages = (0, xmlUtils_js_1.getElementsByTagName)(frame, "draw:image");
485
+ if (drawImages.length > 0) {
486
+ imageHref = drawImages[0].getAttribute("xlink:href") || '';
487
+ if (imageHref) {
488
+ imageHref = cleanAttachmentName(imageHref);
489
+ }
490
+ }
491
+ const imageNode = {
492
+ type: 'image',
493
+ text: '',
494
+ children: [],
495
+ metadata: {
496
+ attachmentName: imageHref,
497
+ ...(altText ? { altText } : {})
498
+ }
499
+ };
500
+ if (config.includeRawContent) {
501
+ imageNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frame, sourceXml, config);
502
+ }
503
+ children.push(imageNode);
361
504
  }
362
- children.push(imageNode);
363
505
  }
364
506
  }
365
507
  }
@@ -713,7 +855,7 @@ const parseOpenOffice = async (buffer, config) => {
713
855
  * @param sourceXml - The source XML string for raw content extraction
714
856
  * @param asSheet - If true, treats tables as sheets (for ODS)
715
857
  */
716
- function traverse(node, targetArray, forceHeading = false, sourceXml, asSheet = false) {
858
+ traverse = (node, targetArray, forceHeading = false, sourceXml = '', asSheet = false) => {
717
859
  if (node.tagName === "text:p") {
718
860
  const pContent = parseParagraphContent(node, paragraphStyleMap, styleMap, config, sourceXml);
719
861
  const type = (forceHeading || (node.getAttribute("text:style-name") || '').toLowerCase().includes('title')) ? 'heading' : 'paragraph';
@@ -1011,8 +1153,7 @@ const parseOpenOffice = async (buffer, config) => {
1011
1153
  // Extract image href to link to attachment
1012
1154
  let imageHref = image.getAttribute("xlink:href") || '';
1013
1155
  if (imageHref) {
1014
- const parts = imageHref.split('/');
1015
- imageHref = parts[parts.length - 1];
1156
+ imageHref = cleanAttachmentName(imageHref);
1016
1157
  }
1017
1158
  const metadata = {
1018
1159
  attachmentName: imageHref,
@@ -1034,25 +1175,47 @@ const parseOpenOffice = async (buffer, config) => {
1034
1175
  targetArray.push(imageNode);
1035
1176
  }
1036
1177
  else if (object) {
1037
- // Handle embedded objects like charts
1178
+ // Handle embedded objects like charts or math formulas
1038
1179
  const href = object.getAttribute("xlink:href");
1039
1180
  if (href) {
1040
- const attachmentName = href.split('/')[0];
1181
+ const attachmentName = cleanAttachmentName(href);
1041
1182
  const objectPath = `${attachmentName}/content.xml`;
1042
1183
  const objectFile = files.find(f => f.path === objectPath || f.path.endsWith(objectPath));
1043
1184
  if (objectFile) {
1044
- const chartData = (0, chartUtils_js_1.extractChartData)(objectFile.content);
1045
- const chartNode = {
1046
- type: 'chart',
1047
- text: chartData.rawTexts.join(" "),
1048
- metadata: {
1049
- attachmentName: attachmentName,
1050
- chartData
1185
+ const objXml = (0, xmlUtils_js_1.parseXmlString)(objectFile.content.toString());
1186
+ const mathNode = (0, xmlUtils_js_1.getFirstElementByTagName)(objXml, "math");
1187
+ if (mathNode) {
1188
+ // Math formula object at block level
1189
+ const formulaText = parseMathML(mathNode).trim();
1190
+ const formulaNode = {
1191
+ type: 'paragraph',
1192
+ text: formulaText,
1193
+ children: [
1194
+ {
1195
+ type: 'text',
1196
+ text: formulaText
1197
+ }
1198
+ ]
1199
+ };
1200
+ if (config.includeRawContent) {
1201
+ formulaNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
1051
1202
  }
1052
- };
1053
- if (config.includeRawContent)
1054
- chartNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
1055
- targetArray.push(chartNode);
1203
+ targetArray.push(formulaNode);
1204
+ }
1205
+ else {
1206
+ const chartData = (0, chartUtils_js_1.extractChartData)(objectFile.content);
1207
+ const chartNode = {
1208
+ type: 'chart',
1209
+ text: chartData.rawTexts.join(" "),
1210
+ metadata: {
1211
+ attachmentName: attachmentName,
1212
+ chartData
1213
+ }
1214
+ };
1215
+ if (config.includeRawContent)
1216
+ chartNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
1217
+ targetArray.push(chartNode);
1218
+ }
1056
1219
  }
1057
1220
  else {
1058
1221
  const chartNode = {
@@ -1077,8 +1240,7 @@ const parseOpenOffice = async (buffer, config) => {
1077
1240
  }
1078
1241
  }
1079
1242
  }
1080
- }
1081
- ;
1243
+ };
1082
1244
  // ODS: Spreadsheet
1083
1245
  if (fileType === 'ods') {
1084
1246
  const spreadsheet = (0, xmlUtils_js_1.getFirstElementByTagName)(body, "office:spreadsheet");
@@ -1156,28 +1318,50 @@ const parseOpenOffice = async (buffer, config) => {
1156
1318
  if (drawImages.length > 0) {
1157
1319
  const rawHref = drawImages[0].getAttribute("xlink:href");
1158
1320
  if (rawHref) {
1159
- const parts = rawHref.split('/');
1160
- imageHref = parts[parts.length - 1];
1321
+ imageHref = cleanAttachmentName(rawHref);
1161
1322
  }
1162
1323
  }
1163
- // Extract chart object href
1324
+ // Extract chart or math object href
1164
1325
  let chartHref = '';
1326
+ let isFormula = false;
1327
+ let formulaText = '';
1165
1328
  const drawObjects = (0, xmlUtils_js_1.getElementsByTagName)(frame, "draw:object");
1166
1329
  if (drawObjects.length > 0) {
1167
1330
  const href = drawObjects[0].getAttribute("xlink:href");
1168
1331
  if (href) {
1169
- // Object href is usually "./Object 1"
1170
- chartHref = href.split('/')[0];
1332
+ chartHref = cleanAttachmentName(href);
1333
+ const objectPath = `${chartHref}/content.xml`;
1334
+ const objectFile = files.find(f => f.path === objectPath || f.path.endsWith(objectPath));
1335
+ if (objectFile) {
1336
+ const objXml = (0, xmlUtils_js_1.parseXmlString)(objectFile.content.toString());
1337
+ const mathNode = (0, xmlUtils_js_1.getFirstElementByTagName)(objXml, "math");
1338
+ if (mathNode) {
1339
+ isFormula = true;
1340
+ formulaText = parseMathML(mathNode).trim();
1341
+ }
1342
+ }
1171
1343
  }
1172
1344
  }
1173
- if (drawImages.length > 0) {
1345
+ if (isFormula) {
1346
+ cellText += formulaText;
1347
+ const textNode = {
1348
+ type: 'text',
1349
+ text: formulaText,
1350
+ formatting: {}
1351
+ };
1352
+ if (config.includeRawContent) {
1353
+ textNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frame, xmlString, config);
1354
+ }
1355
+ children.push(textNode);
1356
+ }
1357
+ else if (drawImages.length > 0) {
1174
1358
  // logic for image node
1175
1359
  const imageNode = {
1176
1360
  type: 'image',
1177
1361
  text: '', // Will be populated by assignAttachmentData
1178
1362
  children: [],
1179
1363
  metadata: {
1180
- attachmentName: imageHref, // Might be empty, will resolve in assignAttachmentData
1364
+ attachmentName: imageHref || chartHref, // Might be empty, will resolve in assignAttachmentData
1181
1365
  ...(altText ? { altText } : {})
1182
1366
  }
1183
1367
  };
@@ -1195,6 +1379,9 @@ const parseOpenOffice = async (buffer, config) => {
1195
1379
  attachmentName: chartHref
1196
1380
  }
1197
1381
  };
1382
+ if (config.includeRawContent) {
1383
+ chartNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frame, xmlString, config);
1384
+ }
1198
1385
  children.push(chartNode);
1199
1386
  }
1200
1387
  }
@@ -63,7 +63,7 @@ const parsePowerPoint = async (buffer, config) => {
63
63
  !!x.match(slideRelsRegex) ||
64
64
  (!config.ignoreComments && (!!x.match(commentsFileRegex) || !!x.match(commentAuthorsRegex))) ||
65
65
  (!config.ignoreSlideMasters && !!x.match(slideMastersRegex)) ||
66
- (!!config.extractAttachments && (!!x.match(mediaFileRegex) || !!x.match(chartFileRegex))));
66
+ (!!config.extractAttachments && (!!x.match(mediaFileRegex) || !!x.match(chartFileRegex))), config.decompressionLimits);
67
67
  // Extract metadata
68
68
  const corePropsFile = files.find(f => f.path.match(corePropsFileRegex));
69
69
  const metadata = corePropsFile ? (0, xmlUtils_js_1.parseOfficeMetadata)(corePropsFile.content.toString()) : {};
@@ -243,7 +243,7 @@ const parseWord = async (buffer, config) => {
243
243
  !!x.match(appPropsFileRegex) ||
244
244
  !!x.match(relsFileRegex) ||
245
245
  !!x.match(stylesFileRegex) ||
246
- (!!config.extractAttachments && !!x.match(mediaFileRegex)));
246
+ (!!config.extractAttachments && !!x.match(mediaFileRegex)), config.decompressionLimits);
247
247
  // Extract metadata
248
248
  const corePropsFile = files.find(f => f.path.match(corePropsFileRegex));
249
249
  const metadata = corePropsFile ? (0, xmlUtils_js_1.parseOfficeMetadata)(corePropsFile.content.toString()) : {};
@@ -473,7 +473,7 @@ const parseWord = async (buffer, config) => {
473
473
  const comments = [];
474
474
  // Traverse children of paragraph (runs, hyperlinks, etc.)
475
475
  const processChildNode = (node) => {
476
- if ((0, xmlUtils_js_1.isElement)(node) && node.nodeName === 'w:r') {
476
+ if ((0, xmlUtils_js_1.isElement)(node) && (node.nodeName === 'w:r' || node.nodeName === 'm:r')) {
477
477
  const runNode = node;
478
478
  const rPr = (0, xmlUtils_js_1.getFirstElementByTagName)(runNode, "w:rPr");
479
479
  // Formatting
@@ -512,7 +512,7 @@ const parseWord = async (buffer, config) => {
512
512
  continue;
513
513
  // also handle unprefixed version (mirroring the behaviour of getElementsByTagName)
514
514
  // Text content
515
- if (child.tagName === "w:t" || child.tagName === "t") {
515
+ if (child.tagName === "w:t" || child.tagName === "t" || child.tagName === "m:t") {
516
516
  const tNode = child;
517
517
  const tContent = tNode.textContent || '';
518
518
  text += tContent;
package/dist/types.d.ts CHANGED
@@ -32,7 +32,15 @@ export declare enum OfficeErrorType {
32
32
  /** Semantic chunking strategy is selected but no embedding function is provided */
33
33
  MISSING_EMBEDDING_FUNCTION = "MISSING_EMBEDDING_FUNCTION",
34
34
  /** The operation was aborted */
35
- OPERATION_ABORTED = "OPERATION_ABORTED"
35
+ OPERATION_ABORTED = "OPERATION_ABORTED",
36
+ /** ZIP entry count exceeds limit */
37
+ ZIP_ENTRY_COUNT_LIMIT_EXCEEDED = "ZIP_ENTRY_COUNT_LIMIT_EXCEEDED",
38
+ /** ZIP entry missing a valid declared size */
39
+ ZIP_ENTRY_INVALID_SIZE = "ZIP_ENTRY_INVALID_SIZE",
40
+ /** ZIP uncompressed size limit exceeded */
41
+ ZIP_SIZE_LIMIT_EXCEEDED = "ZIP_SIZE_LIMIT_EXCEEDED",
42
+ /** Embedding call timed out */
43
+ EMBEDDING_TIMEOUT = "EMBEDDING_TIMEOUT"
36
44
  }
37
45
  /**
38
46
  * Standard warning types for OfficeParser.
@@ -309,7 +317,7 @@ export interface OfficeParserConfig {
309
317
  * The URL/path to the PDF.js worker script.
310
318
  *
311
319
  * **Mandatory** when using PDF parsing in browser environments to avoid worker configuration errors.
312
- * If not provided, it defaults to `https://cdn.jsdelivr.net/npm/pdfjs-dist@5.6.205/build/pdf.worker.min.mjs`.
320
+ * If not provided, it defaults to `https://cdn.jsdelivr.net/npm/pdfjs-dist@6.1.200/build/pdf.worker.min.mjs`.
313
321
  * You can override this with your own local path or a different CDN link.
314
322
  */
315
323
  pdfWorkerSrc?: string;
@@ -347,6 +355,28 @@ export interface OfficeParserConfig {
347
355
  * Defaults to ',' but can be overridden (e.g., ';', '\t').
348
356
  */
349
357
  csvDelimiter?: string;
358
+ /**
359
+ * Limits and checks applied during ZIP extraction to protect against excessive
360
+ * memory and resource usage.
361
+ */
362
+ decompressionLimits?: DecompressionLimits;
363
+ }
364
+ /**
365
+ * Limits applied to ZIP archive decompression.
366
+ */
367
+ export interface DecompressionLimits {
368
+ /**
369
+ * Maximum allowed total uncompressed size (in bytes) of files extracted from a ZIP archive.
370
+ * Applies to OOXML (DOCX, XLSX, PPTX) and ODF (ODT, ODP, ODS) formats.
371
+ * Default is 536870912 (512 MB).
372
+ */
373
+ maxUncompressedBytes?: number;
374
+ /**
375
+ * Maximum allowed number of entries (files and directories) in a ZIP archive.
376
+ * Applies to OOXML (DOCX, XLSX, PPTX) and ODF (ODT, ODP, ODS) formats.
377
+ * Default is 10000.
378
+ */
379
+ maxZipEntries?: number;
350
380
  }
351
381
  /**
352
382
  * A fully-populated parser configuration containing all options.
@@ -1842,4 +1872,7 @@ export interface OfficeParserAST {
1842
1872
  */
1843
1873
  to<T extends this, D extends SupportedDestination<T['type']>>(this: T, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult<D>>;
1844
1874
  }
1875
+ declare global {
1876
+ const __SLIM__: boolean | undefined;
1877
+ }
1845
1878
  export {};
package/dist/types.js CHANGED
@@ -37,6 +37,14 @@ var OfficeErrorType;
37
37
  OfficeErrorType["MISSING_EMBEDDING_FUNCTION"] = "MISSING_EMBEDDING_FUNCTION";
38
38
  /** The operation was aborted */
39
39
  OfficeErrorType["OPERATION_ABORTED"] = "OPERATION_ABORTED";
40
+ /** ZIP entry count exceeds limit */
41
+ OfficeErrorType["ZIP_ENTRY_COUNT_LIMIT_EXCEEDED"] = "ZIP_ENTRY_COUNT_LIMIT_EXCEEDED";
42
+ /** ZIP entry missing a valid declared size */
43
+ OfficeErrorType["ZIP_ENTRY_INVALID_SIZE"] = "ZIP_ENTRY_INVALID_SIZE";
44
+ /** ZIP uncompressed size limit exceeded */
45
+ OfficeErrorType["ZIP_SIZE_LIMIT_EXCEEDED"] = "ZIP_SIZE_LIMIT_EXCEEDED";
46
+ /** Embedding call timed out */
47
+ OfficeErrorType["EMBEDDING_TIMEOUT"] = "EMBEDDING_TIMEOUT";
40
48
  })(OfficeErrorType || (exports.OfficeErrorType = OfficeErrorType = {}));
41
49
  /**
42
50
  * Standard warning types for OfficeParser.
@@ -57,6 +57,12 @@ function isFullParserConfig(config) {
57
57
  */
58
58
  function resolveParserConfig(userConfig) {
59
59
  if (isFullParserConfig(userConfig)) {
60
+ if (!userConfig.decompressionLimits) {
61
+ userConfig.decompressionLimits = {
62
+ maxUncompressedBytes: 512 * 1024 * 1024,
63
+ maxZipEntries: 10000,
64
+ };
65
+ }
60
66
  return userConfig;
61
67
  }
62
68
  // 1. Start with full defaults (deep cloned)
@@ -65,9 +71,15 @@ function resolveParserConfig(userConfig) {
65
71
  return config;
66
72
  }
67
73
  // 2. Merge user config
68
- // We handle ocrConfig specially to avoid shallow-overwriting the whole object
69
- const { ocrConfig, ...rest } = userConfig;
74
+ // We handle ocrConfig and decompressionLimits specially to avoid shallow-overwriting the whole objects
75
+ const { ocrConfig, decompressionLimits, ...rest } = userConfig;
70
76
  Object.assign(config, rest);
77
+ if (decompressionLimits) {
78
+ config.decompressionLimits = {
79
+ ...config.decompressionLimits,
80
+ ...decompressionLimits,
81
+ };
82
+ }
71
83
  if (ocrConfig) {
72
84
  const { timeout, ...ocrRest } = ocrConfig;
73
85
  config.ocrConfig = {
@@ -30,7 +30,11 @@ const ERROR_MESSAGES = {
30
30
  [types_js_1.OfficeErrorType.INVALID_SELECTOR]: (selector) => `Invalid selector: ${selector}`,
31
31
  [types_js_1.OfficeErrorType.INVALID_OUTPUT_MAPPING]: (output) => `Invalid output mapping: ${output}`,
32
32
  [types_js_1.OfficeErrorType.MISSING_EMBEDDING_FUNCTION]: `Semantic chunking requires an "embeddingFunction" to be provided in chunksConfig. This function must accept a string and return a Promise resolving to a number array (vector).`,
33
- [types_js_1.OfficeErrorType.OPERATION_ABORTED]: `The operation was aborted.`
33
+ [types_js_1.OfficeErrorType.OPERATION_ABORTED]: `The operation was aborted.`,
34
+ [types_js_1.OfficeErrorType.ZIP_ENTRY_COUNT_LIMIT_EXCEEDED]: (limit) => `ZIP entry count exceeds limit (${limit})`,
35
+ [types_js_1.OfficeErrorType.ZIP_ENTRY_INVALID_SIZE]: `ZIP entry missing a valid declared size`,
36
+ [types_js_1.OfficeErrorType.ZIP_SIZE_LIMIT_EXCEEDED]: (limit) => `ZIP uncompressed size limit exceeded (${limit} bytes)`,
37
+ [types_js_1.OfficeErrorType.EMBEDDING_TIMEOUT]: (timeout) => `Embedding call timed out after ${timeout}ms`
34
38
  };
35
39
  /**
36
40
  * Lookup table for warning messages.
@@ -13,6 +13,7 @@
13
13
  *
14
14
  * @module zipUtils
15
15
  */
16
+ import { DecompressionLimits } from '../types.js';
16
17
  /**
17
18
  * Represents a file extracted from a ZIP archive.
18
19
  * Contains the file's path within the archive and its content as a Buffer.
@@ -69,5 +70,5 @@ interface ZipFileContent {
69
70
  *
70
71
  * @see https://pkware.cachefly.net/webdocs/casestudies/APPNOTE.TXT ZIP file format specification
71
72
  */
72
- export declare const extractFiles: (zipInput: Buffer, filterFn: (fileName: string) => boolean) => Promise<ZipFileContent[]>;
73
+ export declare const extractFiles: (zipInput: Buffer, filterFn: (fileName: string) => boolean, limits: DecompressionLimits) => Promise<ZipFileContent[]>;
73
74
  export {};
@@ -17,6 +17,8 @@
17
17
  Object.defineProperty(exports, "__esModule", { value: true });
18
18
  exports.extractFiles = void 0;
19
19
  const fflate_1 = require("fflate");
20
+ const types_js_1 = require("../types.js");
21
+ const errorUtils_js_1 = require("./errorUtils.js");
20
22
  /**
21
23
  * Extracts files from a ZIP archive with optional filtering.
22
24
  *
@@ -56,9 +58,39 @@ const fflate_1 = require("fflate");
56
58
  *
57
59
  * @see https://pkware.cachefly.net/webdocs/casestudies/APPNOTE.TXT ZIP file format specification
58
60
  */
59
- const extractFiles = (zipInput, filterFn) => {
61
+ const extractFiles = (zipInput, filterFn, limits) => {
62
+ const maxUncompressedBytes = limits?.maxUncompressedBytes !== undefined && Number.isFinite(limits.maxUncompressedBytes) && limits.maxUncompressedBytes >= 0
63
+ ? limits.maxUncompressedBytes
64
+ : 512 * 1024 * 1024;
65
+ const maxZipEntries = limits?.maxZipEntries !== undefined && Number.isFinite(limits.maxZipEntries) && limits.maxZipEntries >= 0
66
+ ? limits.maxZipEntries
67
+ : 10000;
60
68
  return new Promise((resolve, reject) => {
61
- (0, fflate_1.unzip)(new Uint8Array(zipInput.buffer, zipInput.byteOffset, zipInput.byteLength), { filter: (file) => filterFn(file.name) }, (err, decompressed) => {
69
+ let totalEntryCount = 0;
70
+ let entryCount = 0;
71
+ let declaredTotal = 0;
72
+ (0, fflate_1.unzip)(new Uint8Array(zipInput.buffer, zipInput.byteOffset, zipInput.byteLength), {
73
+ filter: (file) => {
74
+ totalEntryCount++;
75
+ if (totalEntryCount > maxZipEntries) {
76
+ reject((0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.ZIP_ENTRY_COUNT_LIMIT_EXCEEDED, undefined, maxZipEntries));
77
+ return false;
78
+ }
79
+ if (!filterFn(file.name))
80
+ return false;
81
+ if (typeof file.originalSize !== 'number' || !Number.isFinite(file.originalSize) || file.originalSize < 0) {
82
+ reject((0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.ZIP_ENTRY_INVALID_SIZE));
83
+ return false;
84
+ }
85
+ entryCount++;
86
+ declaredTotal += file.originalSize;
87
+ if (declaredTotal > maxUncompressedBytes) {
88
+ reject((0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.ZIP_SIZE_LIMIT_EXCEEDED, undefined, maxUncompressedBytes));
89
+ return false;
90
+ }
91
+ return true;
92
+ }
93
+ }, (err, decompressed) => {
62
94
  if (err)
63
95
  return reject(err);
64
96
  resolve(Object.entries(decompressed).map(([path, data]) => ({