officeparser 7.5.0 → 7.5.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -37,7 +37,7 @@ const parseEpub = async (buffer, config) => {
37
37
  const files = await (0, zipUtils_js_1.extractFiles)(buffer, (path) => /META-INF\/container\.xml$/i.test(path)
38
38
  || /\.opf$/i.test(path)
39
39
  || /\.(xhtml|html|htm)$/i.test(path)
40
- || (!!config.extractAttachments && /\.(png|jpe?g|gif|svg|webp)$/i.test(path)), config.decompressionLimits);
40
+ || (!!config.extractAttachments && /\.(png|jpe?g|gif|svg|webp)$/i.test(path)), config.decompressionLimits, config);
41
41
  // The OPF path is authoritative via META-INF/container.xml; fall back to scanning
42
42
  // for any .opf file for malformed archives that skip the container manifest.
43
43
  let opfPath;
@@ -49,7 +49,7 @@ const parseEpub = async (buffer, config) => {
49
49
  }
50
50
  const opfFile = (opfPath && files.find(f => f.path === opfPath)) || files.find(f => /\.opf$/i.test(f.path));
51
51
  if (!opfFile) {
52
- throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.FILE_CORRUPTED, config, 'epub (no OPF manifest found)');
52
+ throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.REQUIRED_PART_MISSING, config, { fileType: 'epub', part: 'OPF package document (.opf)' });
53
53
  }
54
54
  const opfDir = opfFile.path.includes('/') ? opfFile.path.substring(0, opfFile.path.lastIndexOf('/') + 1) : '';
55
55
  const opfXml = (0, xmlUtils_js_1.parseXmlString)(opfFile.content.toString('utf-8'));
@@ -67,7 +67,16 @@ const parseExcel = async (buffer, config) => {
67
67
  !!x.match(customPropsFileRegex) ||
68
68
  !!x.match(appPropsFileRegex) ||
69
69
  (!!config.extractAttachments && (!!x.match(mediaFileRegex) || !!x.match(drawingRelsRegex))) ||
70
- ((!!config.extractAttachments || !config.ignoreComments) && !!x.match(relsRegex)), config.decompressionLimits);
70
+ ((!!config.extractAttachments || !config.ignoreComments) && !!x.match(relsRegex)), config.decompressionLimits, config);
71
+ // Every workbook has xl/workbook.xml; without it the archive is not a spreadsheet.
72
+ // Resolved up front so a file that cannot be a workbook fails before any of the parsing
73
+ // work below, and read again further down for the sheet-name map.
74
+ const workbookFile = (0, zipUtils_js_1.findRequiredPart)(files, path => path === 'xl/workbook.xml', config, { fileType: 'xlsx', part: 'xl/workbook.xml' });
75
+ // Worksheets, by contrast, are not guaranteed: a workbook holding only chartsheets is
76
+ // valid and simply has no cell text to extract. Warn rather than fail, so the caller can
77
+ // tell "nothing to read here" from "we read nothing".
78
+ if (!files.some(file => !!file.path.match(sheetsRegex)))
79
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.NO_WORKSHEETS_FOUND, config);
71
80
  const sharedStringsFile = files.find(f => f.path === stringsFilePath);
72
81
  // Updated to store structured content (rich text runs) or simple string
73
82
  const sharedStrings = [];
@@ -372,9 +381,8 @@ const parseExcel = async (buffer, config) => {
372
381
  }
373
382
  // Parse workbook.xml to get sheet names and map them to sheet files
374
383
  const sheetNameMap = {};
375
- const workbookFile = files.find(f => f.path === 'xl/workbook.xml');
376
384
  const workbookRelsFile = files.find(f => f.path === 'xl/_rels/workbook.xml.rels');
377
- if (workbookFile && workbookRelsFile) {
385
+ if (workbookRelsFile) {
378
386
  // Parse rels to get rId -> file mapping
379
387
  const relsXml = (0, xmlUtils_js_1.parseXmlString)(workbookRelsFile.content.toString());
380
388
  const relationships = (0, xmlUtils_js_1.getElementsByTagName)(relsXml, "Relationship");
@@ -401,7 +409,6 @@ const parseExcel = async (buffer, config) => {
401
409
  }
402
410
  }
403
411
  const content = [];
404
- const rawContents = [];
405
412
  for (const file of files) {
406
413
  if (file.path.match(mediaFileRegex))
407
414
  continue;
@@ -418,9 +425,6 @@ const parseExcel = async (buffer, config) => {
418
425
  if (file.path.match(drawingRelsRegex))
419
426
  continue;
420
427
  if (file.path.match(sheetsRegex)) {
421
- if (config.includeRawContent) {
422
- rawContents.push(file.content.toString());
423
- }
424
428
  const sheetFilename = file.path.split('/').pop() || '';
425
429
  const relsFilename = `xl/worksheets/_rels/${sheetFilename}.rels`;
426
430
  const relsFile = files.find(f => f.path === relsFilename);
@@ -125,6 +125,8 @@ const cleanAttachmentName = (href) => {
125
125
  const cleaned = href.replace(/^\.\//, '').replace(/\/$/, '');
126
126
  return cleaned.split('/').pop() || '';
127
127
  };
128
+ /** The ODF document types this parser handles, used to validate a caller-supplied file type. */
129
+ const ODF_FILE_TYPES = ['odt', 'odp', 'ods'];
128
130
  /**
129
131
  * Parses an OpenOffice document (.odt, .odp, .ods) and extracts content.
130
132
  *
@@ -148,10 +150,17 @@ const parseOpenOffice = async (buffer, config) => {
148
150
  !!x.match(metaFileRegex) ||
149
151
  !!x.match(stylesFileRegex) ||
150
152
  !!x.match(mimetypeFileRegex) ||
151
- (!!config.extractAttachments && !!x.match(mediaFileRegex)), config.decompressionLimits);
153
+ (!!config.extractAttachments && !!x.match(mediaFileRegex)), config.decompressionLimits, config);
152
154
  // 1. Determine File Type
153
155
  const mimetypeFile = files.find(f => f.path === 'mimetype');
154
- let fileType = 'odt'; // Default
156
+ // The archive's own mimetype entry is authoritative when present. When it is missing,
157
+ // fall back to the type the caller asked for (or that was derived from the extension)
158
+ // rather than assuming text: guessing 'odt' for a spreadsheet sends the parser down the
159
+ // office:text branch, which finds nothing in an office:spreadsheet body and yields an
160
+ // empty document for a perfectly valid file.
161
+ let fileType = ODF_FILE_TYPES.includes(config.fileType)
162
+ ? config.fileType
163
+ : 'odt';
155
164
  if (mimetypeFile) {
156
165
  const mime = mimetypeFile.content.toString().trim();
157
166
  if (mime.includes('spreadsheet'))
@@ -161,7 +170,12 @@ const parseOpenOffice = async (buffer, config) => {
161
170
  else if (mime.includes('text'))
162
171
  fileType = 'odt';
163
172
  }
164
- const mainContentFile = files.find(f => f.path === 'content.xml') || files.find(f => f.path.match(contentFileRegex));
173
+ // The document body is the content.xml at the archive root. The fallback stays anchored
174
+ // and excludes embedded objects: an ODF file can carry Object N/content.xml for a chart
175
+ // or formula, and an unanchored match would promote one of those to the document body
176
+ // when the real one is missing, silently parsing a chart as if it were the whole file.
177
+ const mainContentFile = files.find(f => f.path === 'content.xml')
178
+ || (0, zipUtils_js_1.findRequiredPart)(files, path => /(^|\/)content\.xml$/.test(path) && !objectContentFileRegex.test(path), config, { fileType, part: 'content.xml' });
165
179
  const stylesFile = files.find(f => f.path === 'styles.xml');
166
180
  const stylesDom = stylesFile ? (0, xmlUtils_js_1.parseXmlString)(stylesFile.content.toString()) : undefined;
167
181
  const content = [];
@@ -57,6 +57,7 @@ const parsePowerPoint = async (buffer, config) => {
57
57
  const commentsFileRegex = /ppt\/comments\/comment\d+\.xml/;
58
58
  const commentAuthorsRegex = /ppt\/commentAuthors\.xml/;
59
59
  const slideMastersRegex = /ppt\/slideMasters\/slideMaster\d+\.xml/;
60
+ const presentationFileRegex = /ppt\/presentation\.xml/;
60
61
  const files = await (0, zipUtils_js_1.extractFiles)(buffer, x => !!x.match(config.ignoreNotes ? slidesRegex : allFilesRegex) ||
61
62
  !!x.match(corePropsFileRegex) ||
62
63
  !!x.match(customPropsFileRegex) ||
@@ -64,7 +65,14 @@ const parsePowerPoint = async (buffer, config) => {
64
65
  !!x.match(slideRelsRegex) ||
65
66
  (!config.ignoreComments && (!!x.match(commentsFileRegex) || !!x.match(commentAuthorsRegex))) ||
66
67
  (!config.ignoreSlideMasters && !!x.match(slideMastersRegex)) ||
67
- (!!config.extractAttachments && (!!x.match(mediaFileRegex) || !!x.match(chartFileRegex))), config.decompressionLimits);
68
+ !!x.match(presentationFileRegex) ||
69
+ (!!config.extractAttachments && (!!x.match(mediaFileRegex) || !!x.match(chartFileRegex))), config.decompressionLimits, config);
70
+ // ppt/presentation.xml is the part that makes an archive a presentation, and unlike the
71
+ // slides it is always present: PowerPoint can save a deck with no slides at all, so an
72
+ // empty ppt/slides/ is a warning rather than a failure.
73
+ (0, zipUtils_js_1.findRequiredPart)(files, path => !!path.match(presentationFileRegex), config, { fileType: 'pptx', part: 'ppt/presentation.xml' });
74
+ if (!files.some(file => !!file.path.match(slidesRegex)))
75
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.NO_SLIDES_FOUND, config);
68
76
  // Extract metadata
69
77
  const corePropsFile = files.find(f => f.path.match(corePropsFileRegex));
70
78
  const metadata = corePropsFile ? (0, xmlUtils_js_1.parseOfficeMetadata)(corePropsFile.content.toString()) : {};
@@ -89,7 +97,6 @@ const parsePowerPoint = async (buffer, config) => {
89
97
  return aNum - bNum;
90
98
  });
91
99
  const content = [];
92
- const rawContents = [];
93
100
  const slideRelsMap = {};
94
101
  const authorMap = {};
95
102
  if (!config.ignoreComments) {
@@ -752,15 +759,24 @@ const parsePowerPoint = async (buffer, config) => {
752
759
  continue;
753
760
  if (file.path.match(slideRelsRegex))
754
761
  continue;
762
+ // All three document-property parts are extracted for metadata and were read earlier;
763
+ // only the core one was skipped here, leaving the other two to be parsed again as if
764
+ // they might be slides.
755
765
  if (file.path.match(corePropsFileRegex))
756
766
  continue;
767
+ if (file.path.match(appPropsFileRegex))
768
+ continue;
769
+ if (file.path.match(customPropsFileRegex))
770
+ continue;
757
771
  if (file.path.includes("comment"))
758
772
  continue;
773
+ // This loop treats every remaining file as a slide or note, so the presentation part
774
+ // has to be skipped explicitly: it carries no slide number and would otherwise be
775
+ // added to the deck as an empty slide.
776
+ if (file.path.match(presentationFileRegex))
777
+ continue;
759
778
  const xmlContentString = file.content.toString();
760
779
  const xml = (0, xmlUtils_js_1.parseXmlString)(xmlContentString, { locator: config.includeRawContent });
761
- if (config.includeRawContent) {
762
- rawContents.push(xmlContentString);
763
- }
764
780
  const slideMatch = file.path.match(slideNumberRegex);
765
781
  const slideNumber = slideMatch ? parseInt(slideMatch[1]) : 0;
766
782
  const masterMatch = file.path.match(/slideMaster(\d+)\.xml/);
@@ -95,8 +95,12 @@ const parseWord = async (buffer, config) => {
95
95
  const footnotesFileRegex = /word\/footnotes[\d+]?.xml/;
96
96
  const endnotesFileRegex = /word\/endnotes[\d+]?.xml/;
97
97
  const commentsFileRegex = /word\/comments[\d+]?.xml/;
98
- const headerFileRegex = /word\/header[\d+]?.xml/;
99
- const footerFileRegex = /word\/footer[\d+]?.xml/;
98
+ // Headers and footers are the only parts a document can have many of: Word writes up to
99
+ // three per section (default, first page, even pages), so a handful of sections is enough
100
+ // to reach header10.xml. The single-character form the other parts use stops matching at
101
+ // nine, which would drop those later files as silently as not extracting them at all.
102
+ const headerFileRegex = /word\/header\d*\.xml/;
103
+ const footerFileRegex = /word\/footer\d*\.xml/;
100
104
  const numberingFileRegex = /word\/numbering[\d+]?.xml/;
101
105
  const mediaFileRegex = /(word\/)?media\/.*/;
102
106
  const corePropsFileRegex = /docProps\/core[\d+]?.xml/;
@@ -244,7 +248,12 @@ const parseWord = async (buffer, config) => {
244
248
  !!x.match(appPropsFileRegex) ||
245
249
  !!x.match(relsFileRegex) ||
246
250
  !!x.match(stylesFileRegex) ||
247
- (!!config.extractAttachments && !!x.match(mediaFileRegex)), config.decompressionLimits);
251
+ (!config.ignoreComments && !!x.match(commentsFileRegex)) ||
252
+ (!config.ignoreHeadersAndFooters && (!!x.match(headerFileRegex) || !!x.match(footerFileRegex))) ||
253
+ (!!config.extractAttachments && !!x.match(mediaFileRegex)), config.decompressionLimits, config);
254
+ // A DOCX without its main document part is not a DOCX. Checked with the same regex the
255
+ // parse loop below uses to recognize it, so the two cannot fall out of step.
256
+ (0, zipUtils_js_1.findRequiredPart)(files, path => !!path.match(documentFileRegex), config, { fileType: 'docx', part: 'word/document.xml' });
248
257
  // Extract metadata
249
258
  const corePropsFile = files.find(f => f.path.match(corePropsFileRegex));
250
259
  const metadata = corePropsFile ? (0, xmlUtils_js_1.parseOfficeMetadata)(corePropsFile.content.toString()) : {};
@@ -405,7 +414,6 @@ const parseWord = async (buffer, config) => {
405
414
  }
406
415
  }
407
416
  const content = [];
408
- const rawContents = [];
409
417
  const numberingState = {};
410
418
  const listCounters = {}; // Track item index per listId/level
411
419
  // Helper to parse a paragraph node
@@ -1092,9 +1100,6 @@ const parseWord = async (buffer, config) => {
1092
1100
  if (file.path.match(footerFileRegex))
1093
1101
  continue;
1094
1102
  const documentContent = file.content.toString();
1095
- if (config.includeRawContent) {
1096
- rawContents.push(documentContent);
1097
- }
1098
1103
  const doc = (0, xmlUtils_js_1.parseXmlString)(documentContent, { locator: config.includeRawContent });
1099
1104
  const body = (0, xmlUtils_js_1.getFirstElementByTagName)(doc, "w:body");
1100
1105
  if (body) {