officeparser 7.4.0 → 7.5.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. package/README.md +85 -4
  2. package/dist/OfficeParser.js +54 -6
  3. package/dist/generators/BaseGenerator.d.ts +15 -0
  4. package/dist/generators/BaseGenerator.js +31 -0
  5. package/dist/generators/HtmlGenerator.d.ts +9 -0
  6. package/dist/generators/HtmlGenerator.js +34 -4
  7. package/dist/generators/MarkdownGenerator.d.ts +13 -0
  8. package/dist/generators/MarkdownGenerator.js +115 -41
  9. package/dist/generators/RtfGenerator.d.ts +13 -0
  10. package/dist/generators/RtfGenerator.js +23 -2
  11. package/dist/index.d.ts +2 -2
  12. package/dist/officeparser.browser.d.ts +34 -1
  13. package/dist/officeparser.browser.iife.js +160 -160
  14. package/dist/officeparser.browser.mjs +198 -198
  15. package/dist/officeparser.browser.slim.d.ts +34 -1
  16. package/dist/officeparser.browser.slim.iife.js +186 -186
  17. package/dist/officeparser.browser.slim.mjs +186 -186
  18. package/dist/parsers/EpubParser.js +2 -2
  19. package/dist/parsers/ExcelParser.js +11 -7
  20. package/dist/parsers/HtmlParser.js +39 -0
  21. package/dist/parsers/OpenOfficeParser.js +139 -167
  22. package/dist/parsers/PowerPointParser.js +48 -11
  23. package/dist/parsers/WordParser.js +33 -7
  24. package/dist/sbom.cdx.json +92 -92
  25. package/dist/types.d.ts +34 -1
  26. package/dist/types.js +10 -0
  27. package/dist/utils/configUtils.d.ts +15 -2
  28. package/dist/utils/configUtils.js +58 -13
  29. package/dist/utils/errorUtils.d.ts +8 -2
  30. package/dist/utils/errorUtils.js +23 -1
  31. package/dist/utils/mathUtils.d.ts +42 -0
  32. package/dist/utils/mathUtils.js +385 -0
  33. package/dist/utils/zipUtils.d.ts +64 -4
  34. package/dist/utils/zipUtils.js +188 -4
  35. package/package.json +9 -5
@@ -29,6 +29,7 @@ const astUtils_js_1 = require("../utils/astUtils.js");
29
29
  const chartUtils_js_1 = require("../utils/chartUtils.js");
30
30
  const errorUtils_js_1 = require("../utils/errorUtils.js");
31
31
  const imageUtils_js_1 = require("../utils/imageUtils.js");
32
+ const mathUtils_js_1 = require("../utils/mathUtils.js");
32
33
  const ocrUtils_js_1 = require("../utils/ocrUtils.js");
33
34
  const xmlUtils_js_1 = require("../utils/xmlUtils.js");
34
35
  const zipUtils_js_1 = require("../utils/zipUtils.js");
@@ -56,6 +57,7 @@ const parsePowerPoint = async (buffer, config) => {
56
57
  const commentsFileRegex = /ppt\/comments\/comment\d+\.xml/;
57
58
  const commentAuthorsRegex = /ppt\/commentAuthors\.xml/;
58
59
  const slideMastersRegex = /ppt\/slideMasters\/slideMaster\d+\.xml/;
60
+ const presentationFileRegex = /ppt\/presentation\.xml/;
59
61
  const files = await (0, zipUtils_js_1.extractFiles)(buffer, x => !!x.match(config.ignoreNotes ? slidesRegex : allFilesRegex) ||
60
62
  !!x.match(corePropsFileRegex) ||
61
63
  !!x.match(customPropsFileRegex) ||
@@ -63,7 +65,14 @@ const parsePowerPoint = async (buffer, config) => {
63
65
  !!x.match(slideRelsRegex) ||
64
66
  (!config.ignoreComments && (!!x.match(commentsFileRegex) || !!x.match(commentAuthorsRegex))) ||
65
67
  (!config.ignoreSlideMasters && !!x.match(slideMastersRegex)) ||
66
- (!!config.extractAttachments && (!!x.match(mediaFileRegex) || !!x.match(chartFileRegex))), config.decompressionLimits);
68
+ !!x.match(presentationFileRegex) ||
69
+ (!!config.extractAttachments && (!!x.match(mediaFileRegex) || !!x.match(chartFileRegex))), config.decompressionLimits, config);
70
+ // ppt/presentation.xml is the part that makes an archive a presentation, and unlike the
71
+ // slides it is always present: PowerPoint can save a deck with no slides at all, so an
72
+ // empty ppt/slides/ is a warning rather than a failure.
73
+ (0, zipUtils_js_1.findRequiredPart)(files, path => !!path.match(presentationFileRegex), config, { fileType: 'pptx', part: 'ppt/presentation.xml' });
74
+ if (!files.some(file => !!file.path.match(slidesRegex)))
75
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.NO_SLIDES_FOUND, config);
67
76
  // Extract metadata
68
77
  const corePropsFile = files.find(f => f.path.match(corePropsFileRegex));
69
78
  const metadata = corePropsFile ? (0, xmlUtils_js_1.parseOfficeMetadata)(corePropsFile.content.toString()) : {};
@@ -88,7 +97,6 @@ const parsePowerPoint = async (buffer, config) => {
88
97
  return aNum - bNum;
89
98
  });
90
99
  const content = [];
91
- const rawContents = [];
92
100
  const slideRelsMap = {};
93
101
  const authorMap = {};
94
102
  if (!config.ignoreComments) {
@@ -576,6 +584,30 @@ const parsePowerPoint = async (buffer, config) => {
576
584
  activeNode.children?.push({ type: 'text', text: "\n" });
577
585
  }
578
586
  }
587
+ else {
588
+ // Equations. This loop dispatches on `a:r`/`a:fld`, so an `m:oMath` -
589
+ // which is a sibling of the runs, not one of them - was never visited at
590
+ // all and the formula vanished from the slide without a warning.
591
+ //
592
+ // PowerPoint writes the equation either directly in the paragraph or
593
+ // wrapped in `mc:AlternateContent`/`a14:m` for pre-2010 readers, so take
594
+ // the element itself when it is the equation and search inside it
595
+ // otherwise. `getElementsByTagName` returns document order, which is the
596
+ // order the equations are read in.
597
+ const isMath = tag === "m:oMath" || tag === "m:oMathPara";
598
+ const equations = isMath ? [element] : (0, xmlUtils_js_1.getElementsByTagName)(element, "m:oMath");
599
+ for (const equation of equations) {
600
+ const latex = (0, mathUtils_js_1.ommlToLatex)(equation);
601
+ if ((0, mathUtils_js_1.isEmptyMath)(latex))
602
+ continue;
603
+ activeNode.text += latex;
604
+ activeNode.children?.push({
605
+ type: 'code',
606
+ text: latex,
607
+ metadata: { math: tag === "m:oMathPara" ? 'block' : 'inline' }
608
+ });
609
+ }
610
+ }
579
611
  }
580
612
  }
581
613
  }
@@ -623,12 +655,8 @@ const parsePowerPoint = async (buffer, config) => {
623
655
  }
624
656
  // Case 4: Grouped shape (recursive!)
625
657
  else if (tag === "p:grpSp") {
626
- // Extract the nested <p:spTree> inside the group
627
- const nestedTree = (0, xmlUtils_js_1.getFirstElementByTagName)(element, "p:spTree");
628
- // Recurse into the nested tree
629
- if (nestedTree) {
630
- nodes.push(...traverseSpTree(nestedTree, slideNumber, xmlContentString));
631
- }
658
+ // Recurse into the group element itself which holds the child shapes
659
+ nodes.push(...traverseSpTree(element, slideNumber, xmlContentString));
632
660
  }
633
661
  }
634
662
  return nodes;
@@ -731,15 +759,24 @@ const parsePowerPoint = async (buffer, config) => {
731
759
  continue;
732
760
  if (file.path.match(slideRelsRegex))
733
761
  continue;
762
+ // All three document-property parts are extracted for metadata and were read earlier;
763
+ // only the core one was skipped here, leaving the other two to be parsed again as if
764
+ // they might be slides.
734
765
  if (file.path.match(corePropsFileRegex))
735
766
  continue;
767
+ if (file.path.match(appPropsFileRegex))
768
+ continue;
769
+ if (file.path.match(customPropsFileRegex))
770
+ continue;
736
771
  if (file.path.includes("comment"))
737
772
  continue;
773
+ // This loop treats every remaining file as a slide or note, so the presentation part
774
+ // has to be skipped explicitly: it carries no slide number and would otherwise be
775
+ // added to the deck as an empty slide.
776
+ if (file.path.match(presentationFileRegex))
777
+ continue;
738
778
  const xmlContentString = file.content.toString();
739
779
  const xml = (0, xmlUtils_js_1.parseXmlString)(xmlContentString, { locator: config.includeRawContent });
740
- if (config.includeRawContent) {
741
- rawContents.push(xmlContentString);
742
- }
743
780
  const slideMatch = file.path.match(slideNumberRegex);
744
781
  const slideNumber = slideMatch ? parseInt(slideMatch[1]) : 0;
745
782
  const masterMatch = file.path.match(/slideMaster(\d+)\.xml/);
@@ -66,6 +66,7 @@ const types_js_1 = require("../types.js");
66
66
  const astUtils_js_1 = require("../utils/astUtils.js");
67
67
  const errorUtils_js_1 = require("../utils/errorUtils.js");
68
68
  const imageUtils_js_1 = require("../utils/imageUtils.js");
69
+ const mathUtils_js_1 = require("../utils/mathUtils.js");
69
70
  const ocrUtils_js_1 = require("../utils/ocrUtils.js");
70
71
  const xmlUtils_js_1 = require("../utils/xmlUtils.js");
71
72
  const zipUtils_js_1 = require("../utils/zipUtils.js");
@@ -94,8 +95,12 @@ const parseWord = async (buffer, config) => {
94
95
  const footnotesFileRegex = /word\/footnotes[\d+]?.xml/;
95
96
  const endnotesFileRegex = /word\/endnotes[\d+]?.xml/;
96
97
  const commentsFileRegex = /word\/comments[\d+]?.xml/;
97
- const headerFileRegex = /word\/header[\d+]?.xml/;
98
- const footerFileRegex = /word\/footer[\d+]?.xml/;
98
+ // Headers and footers are the only parts a document can have many of: Word writes up to
99
+ // three per section (default, first page, even pages), so a handful of sections is enough
100
+ // to reach header10.xml. The single-character form the other parts use stops matching at
101
+ // nine, which would drop those later files as silently as not extracting them at all.
102
+ const headerFileRegex = /word\/header\d*\.xml/;
103
+ const footerFileRegex = /word\/footer\d*\.xml/;
99
104
  const numberingFileRegex = /word\/numbering[\d+]?.xml/;
100
105
  const mediaFileRegex = /(word\/)?media\/.*/;
101
106
  const corePropsFileRegex = /docProps\/core[\d+]?.xml/;
@@ -243,7 +248,12 @@ const parseWord = async (buffer, config) => {
243
248
  !!x.match(appPropsFileRegex) ||
244
249
  !!x.match(relsFileRegex) ||
245
250
  !!x.match(stylesFileRegex) ||
246
- (!!config.extractAttachments && !!x.match(mediaFileRegex)), config.decompressionLimits);
251
+ (!config.ignoreComments && !!x.match(commentsFileRegex)) ||
252
+ (!config.ignoreHeadersAndFooters && (!!x.match(headerFileRegex) || !!x.match(footerFileRegex))) ||
253
+ (!!config.extractAttachments && !!x.match(mediaFileRegex)), config.decompressionLimits, config);
254
+ // A DOCX without its main document part is not a DOCX. Checked with the same regex the
255
+ // parse loop below uses to recognize it, so the two cannot fall out of step.
256
+ (0, zipUtils_js_1.findRequiredPart)(files, path => !!path.match(documentFileRegex), config, { fileType: 'docx', part: 'word/document.xml' });
247
257
  // Extract metadata
248
258
  const corePropsFile = files.find(f => f.path.match(corePropsFileRegex));
249
259
  const metadata = corePropsFile ? (0, xmlUtils_js_1.parseOfficeMetadata)(corePropsFile.content.toString()) : {};
@@ -404,7 +414,6 @@ const parseWord = async (buffer, config) => {
404
414
  }
405
415
  }
406
416
  const content = [];
407
- const rawContents = [];
408
417
  const numberingState = {};
409
418
  const listCounters = {}; // Track item index per listId/level
410
419
  // Helper to parse a paragraph node
@@ -751,6 +760,26 @@ const parseWord = async (buffer, config) => {
751
760
  }
752
761
  }
753
762
  }
763
+ else if ((0, xmlUtils_js_1.isElement)(node) && (node.nodeName === 'm:oMath' || node.nodeName === 'oMath'
764
+ || node.nodeName === 'm:oMathPara' || node.nodeName === 'oMathPara')) {
765
+ // Equations. Without this branch they reach the generic fallback below, which
766
+ // recurses into every child and concatenates the `m:t` runs with no separators -
767
+ // so `<m:num>1</m:num><m:den>2</m:den>` came out as "12". That is worse than
768
+ // dropping the formula: the result still reads as a number, so nothing downstream
769
+ // can tell it is wrong.
770
+ //
771
+ // `m:oMathPara` is a display equation on its own line; a bare `m:oMath` is inline.
772
+ const isBlock = node.nodeName === 'm:oMathPara' || node.nodeName === 'oMathPara';
773
+ const latex = (0, mathUtils_js_1.ommlToLatex)(node);
774
+ if (!(0, mathUtils_js_1.isEmptyMath)(latex)) {
775
+ text += latex;
776
+ children.push({
777
+ type: 'code',
778
+ text: latex,
779
+ metadata: { math: isBlock ? 'block' : 'inline' }
780
+ });
781
+ }
782
+ }
754
783
  else if (node.childNodes.length > 0) {
755
784
  // Generic fallback for unknown elements that might contain content
756
785
  for (const child of Array.from(node.childNodes))
@@ -1071,9 +1100,6 @@ const parseWord = async (buffer, config) => {
1071
1100
  if (file.path.match(footerFileRegex))
1072
1101
  continue;
1073
1102
  const documentContent = file.content.toString();
1074
- if (config.includeRawContent) {
1075
- rawContents.push(documentContent);
1076
- }
1077
1103
  const doc = (0, xmlUtils_js_1.parseXmlString)(documentContent, { locator: config.includeRawContent });
1078
1104
  const body = (0, xmlUtils_js_1.getFirstElementByTagName)(doc, "w:body");
1079
1105
  if (body) {