officeparser 7.4.0 → 7.5.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +85 -4
- package/dist/OfficeParser.js +54 -6
- package/dist/generators/BaseGenerator.d.ts +15 -0
- package/dist/generators/BaseGenerator.js +31 -0
- package/dist/generators/HtmlGenerator.d.ts +9 -0
- package/dist/generators/HtmlGenerator.js +34 -4
- package/dist/generators/MarkdownGenerator.d.ts +13 -0
- package/dist/generators/MarkdownGenerator.js +115 -41
- package/dist/generators/RtfGenerator.d.ts +13 -0
- package/dist/generators/RtfGenerator.js +23 -2
- package/dist/index.d.ts +2 -2
- package/dist/officeparser.browser.d.ts +34 -1
- package/dist/officeparser.browser.iife.js +160 -160
- package/dist/officeparser.browser.mjs +198 -198
- package/dist/officeparser.browser.slim.d.ts +34 -1
- package/dist/officeparser.browser.slim.iife.js +186 -186
- package/dist/officeparser.browser.slim.mjs +186 -186
- package/dist/parsers/EpubParser.js +2 -2
- package/dist/parsers/ExcelParser.js +11 -7
- package/dist/parsers/HtmlParser.js +39 -0
- package/dist/parsers/OpenOfficeParser.js +139 -167
- package/dist/parsers/PowerPointParser.js +48 -11
- package/dist/parsers/WordParser.js +33 -7
- package/dist/sbom.cdx.json +92 -92
- package/dist/types.d.ts +34 -1
- package/dist/types.js +10 -0
- package/dist/utils/configUtils.d.ts +15 -2
- package/dist/utils/configUtils.js +58 -13
- package/dist/utils/errorUtils.d.ts +8 -2
- package/dist/utils/errorUtils.js +23 -1
- package/dist/utils/mathUtils.d.ts +42 -0
- package/dist/utils/mathUtils.js +385 -0
- package/dist/utils/zipUtils.d.ts +64 -4
- package/dist/utils/zipUtils.js +188 -4
- package/package.json +9 -5
|
@@ -29,6 +29,7 @@ const astUtils_js_1 = require("../utils/astUtils.js");
|
|
|
29
29
|
const chartUtils_js_1 = require("../utils/chartUtils.js");
|
|
30
30
|
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
31
31
|
const imageUtils_js_1 = require("../utils/imageUtils.js");
|
|
32
|
+
const mathUtils_js_1 = require("../utils/mathUtils.js");
|
|
32
33
|
const ocrUtils_js_1 = require("../utils/ocrUtils.js");
|
|
33
34
|
const xmlUtils_js_1 = require("../utils/xmlUtils.js");
|
|
34
35
|
const zipUtils_js_1 = require("../utils/zipUtils.js");
|
|
@@ -56,6 +57,7 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
56
57
|
const commentsFileRegex = /ppt\/comments\/comment\d+\.xml/;
|
|
57
58
|
const commentAuthorsRegex = /ppt\/commentAuthors\.xml/;
|
|
58
59
|
const slideMastersRegex = /ppt\/slideMasters\/slideMaster\d+\.xml/;
|
|
60
|
+
const presentationFileRegex = /ppt\/presentation\.xml/;
|
|
59
61
|
const files = await (0, zipUtils_js_1.extractFiles)(buffer, x => !!x.match(config.ignoreNotes ? slidesRegex : allFilesRegex) ||
|
|
60
62
|
!!x.match(corePropsFileRegex) ||
|
|
61
63
|
!!x.match(customPropsFileRegex) ||
|
|
@@ -63,7 +65,14 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
63
65
|
!!x.match(slideRelsRegex) ||
|
|
64
66
|
(!config.ignoreComments && (!!x.match(commentsFileRegex) || !!x.match(commentAuthorsRegex))) ||
|
|
65
67
|
(!config.ignoreSlideMasters && !!x.match(slideMastersRegex)) ||
|
|
66
|
-
|
|
68
|
+
!!x.match(presentationFileRegex) ||
|
|
69
|
+
(!!config.extractAttachments && (!!x.match(mediaFileRegex) || !!x.match(chartFileRegex))), config.decompressionLimits, config);
|
|
70
|
+
// ppt/presentation.xml is the part that makes an archive a presentation, and unlike the
|
|
71
|
+
// slides it is always present: PowerPoint can save a deck with no slides at all, so an
|
|
72
|
+
// empty ppt/slides/ is a warning rather than a failure.
|
|
73
|
+
(0, zipUtils_js_1.findRequiredPart)(files, path => !!path.match(presentationFileRegex), config, { fileType: 'pptx', part: 'ppt/presentation.xml' });
|
|
74
|
+
if (!files.some(file => !!file.path.match(slidesRegex)))
|
|
75
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.NO_SLIDES_FOUND, config);
|
|
67
76
|
// Extract metadata
|
|
68
77
|
const corePropsFile = files.find(f => f.path.match(corePropsFileRegex));
|
|
69
78
|
const metadata = corePropsFile ? (0, xmlUtils_js_1.parseOfficeMetadata)(corePropsFile.content.toString()) : {};
|
|
@@ -88,7 +97,6 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
88
97
|
return aNum - bNum;
|
|
89
98
|
});
|
|
90
99
|
const content = [];
|
|
91
|
-
const rawContents = [];
|
|
92
100
|
const slideRelsMap = {};
|
|
93
101
|
const authorMap = {};
|
|
94
102
|
if (!config.ignoreComments) {
|
|
@@ -576,6 +584,30 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
576
584
|
activeNode.children?.push({ type: 'text', text: "\n" });
|
|
577
585
|
}
|
|
578
586
|
}
|
|
587
|
+
else {
|
|
588
|
+
// Equations. This loop dispatches on `a:r`/`a:fld`, so an `m:oMath` -
|
|
589
|
+
// which is a sibling of the runs, not one of them - was never visited at
|
|
590
|
+
// all and the formula vanished from the slide without a warning.
|
|
591
|
+
//
|
|
592
|
+
// PowerPoint writes the equation either directly in the paragraph or
|
|
593
|
+
// wrapped in `mc:AlternateContent`/`a14:m` for pre-2010 readers, so take
|
|
594
|
+
// the element itself when it is the equation and search inside it
|
|
595
|
+
// otherwise. `getElementsByTagName` returns document order, which is the
|
|
596
|
+
// order the equations are read in.
|
|
597
|
+
const isMath = tag === "m:oMath" || tag === "m:oMathPara";
|
|
598
|
+
const equations = isMath ? [element] : (0, xmlUtils_js_1.getElementsByTagName)(element, "m:oMath");
|
|
599
|
+
for (const equation of equations) {
|
|
600
|
+
const latex = (0, mathUtils_js_1.ommlToLatex)(equation);
|
|
601
|
+
if ((0, mathUtils_js_1.isEmptyMath)(latex))
|
|
602
|
+
continue;
|
|
603
|
+
activeNode.text += latex;
|
|
604
|
+
activeNode.children?.push({
|
|
605
|
+
type: 'code',
|
|
606
|
+
text: latex,
|
|
607
|
+
metadata: { math: tag === "m:oMathPara" ? 'block' : 'inline' }
|
|
608
|
+
});
|
|
609
|
+
}
|
|
610
|
+
}
|
|
579
611
|
}
|
|
580
612
|
}
|
|
581
613
|
}
|
|
@@ -623,12 +655,8 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
623
655
|
}
|
|
624
656
|
// Case 4: Grouped shape (recursive!)
|
|
625
657
|
else if (tag === "p:grpSp") {
|
|
626
|
-
//
|
|
627
|
-
|
|
628
|
-
// Recurse into the nested tree
|
|
629
|
-
if (nestedTree) {
|
|
630
|
-
nodes.push(...traverseSpTree(nestedTree, slideNumber, xmlContentString));
|
|
631
|
-
}
|
|
658
|
+
// Recurse into the group element itself which holds the child shapes
|
|
659
|
+
nodes.push(...traverseSpTree(element, slideNumber, xmlContentString));
|
|
632
660
|
}
|
|
633
661
|
}
|
|
634
662
|
return nodes;
|
|
@@ -731,15 +759,24 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
731
759
|
continue;
|
|
732
760
|
if (file.path.match(slideRelsRegex))
|
|
733
761
|
continue;
|
|
762
|
+
// All three document-property parts are extracted for metadata and were read earlier;
|
|
763
|
+
// only the core one was skipped here, leaving the other two to be parsed again as if
|
|
764
|
+
// they might be slides.
|
|
734
765
|
if (file.path.match(corePropsFileRegex))
|
|
735
766
|
continue;
|
|
767
|
+
if (file.path.match(appPropsFileRegex))
|
|
768
|
+
continue;
|
|
769
|
+
if (file.path.match(customPropsFileRegex))
|
|
770
|
+
continue;
|
|
736
771
|
if (file.path.includes("comment"))
|
|
737
772
|
continue;
|
|
773
|
+
// This loop treats every remaining file as a slide or note, so the presentation part
|
|
774
|
+
// has to be skipped explicitly: it carries no slide number and would otherwise be
|
|
775
|
+
// added to the deck as an empty slide.
|
|
776
|
+
if (file.path.match(presentationFileRegex))
|
|
777
|
+
continue;
|
|
738
778
|
const xmlContentString = file.content.toString();
|
|
739
779
|
const xml = (0, xmlUtils_js_1.parseXmlString)(xmlContentString, { locator: config.includeRawContent });
|
|
740
|
-
if (config.includeRawContent) {
|
|
741
|
-
rawContents.push(xmlContentString);
|
|
742
|
-
}
|
|
743
780
|
const slideMatch = file.path.match(slideNumberRegex);
|
|
744
781
|
const slideNumber = slideMatch ? parseInt(slideMatch[1]) : 0;
|
|
745
782
|
const masterMatch = file.path.match(/slideMaster(\d+)\.xml/);
|
|
@@ -66,6 +66,7 @@ const types_js_1 = require("../types.js");
|
|
|
66
66
|
const astUtils_js_1 = require("../utils/astUtils.js");
|
|
67
67
|
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
68
68
|
const imageUtils_js_1 = require("../utils/imageUtils.js");
|
|
69
|
+
const mathUtils_js_1 = require("../utils/mathUtils.js");
|
|
69
70
|
const ocrUtils_js_1 = require("../utils/ocrUtils.js");
|
|
70
71
|
const xmlUtils_js_1 = require("../utils/xmlUtils.js");
|
|
71
72
|
const zipUtils_js_1 = require("../utils/zipUtils.js");
|
|
@@ -94,8 +95,12 @@ const parseWord = async (buffer, config) => {
|
|
|
94
95
|
const footnotesFileRegex = /word\/footnotes[\d+]?.xml/;
|
|
95
96
|
const endnotesFileRegex = /word\/endnotes[\d+]?.xml/;
|
|
96
97
|
const commentsFileRegex = /word\/comments[\d+]?.xml/;
|
|
97
|
-
|
|
98
|
-
|
|
98
|
+
// Headers and footers are the only parts a document can have many of: Word writes up to
|
|
99
|
+
// three per section (default, first page, even pages), so a handful of sections is enough
|
|
100
|
+
// to reach header10.xml. The single-character form the other parts use stops matching at
|
|
101
|
+
// nine, which would drop those later files as silently as not extracting them at all.
|
|
102
|
+
const headerFileRegex = /word\/header\d*\.xml/;
|
|
103
|
+
const footerFileRegex = /word\/footer\d*\.xml/;
|
|
99
104
|
const numberingFileRegex = /word\/numbering[\d+]?.xml/;
|
|
100
105
|
const mediaFileRegex = /(word\/)?media\/.*/;
|
|
101
106
|
const corePropsFileRegex = /docProps\/core[\d+]?.xml/;
|
|
@@ -243,7 +248,12 @@ const parseWord = async (buffer, config) => {
|
|
|
243
248
|
!!x.match(appPropsFileRegex) ||
|
|
244
249
|
!!x.match(relsFileRegex) ||
|
|
245
250
|
!!x.match(stylesFileRegex) ||
|
|
246
|
-
(
|
|
251
|
+
(!config.ignoreComments && !!x.match(commentsFileRegex)) ||
|
|
252
|
+
(!config.ignoreHeadersAndFooters && (!!x.match(headerFileRegex) || !!x.match(footerFileRegex))) ||
|
|
253
|
+
(!!config.extractAttachments && !!x.match(mediaFileRegex)), config.decompressionLimits, config);
|
|
254
|
+
// A DOCX without its main document part is not a DOCX. Checked with the same regex the
|
|
255
|
+
// parse loop below uses to recognize it, so the two cannot fall out of step.
|
|
256
|
+
(0, zipUtils_js_1.findRequiredPart)(files, path => !!path.match(documentFileRegex), config, { fileType: 'docx', part: 'word/document.xml' });
|
|
247
257
|
// Extract metadata
|
|
248
258
|
const corePropsFile = files.find(f => f.path.match(corePropsFileRegex));
|
|
249
259
|
const metadata = corePropsFile ? (0, xmlUtils_js_1.parseOfficeMetadata)(corePropsFile.content.toString()) : {};
|
|
@@ -404,7 +414,6 @@ const parseWord = async (buffer, config) => {
|
|
|
404
414
|
}
|
|
405
415
|
}
|
|
406
416
|
const content = [];
|
|
407
|
-
const rawContents = [];
|
|
408
417
|
const numberingState = {};
|
|
409
418
|
const listCounters = {}; // Track item index per listId/level
|
|
410
419
|
// Helper to parse a paragraph node
|
|
@@ -751,6 +760,26 @@ const parseWord = async (buffer, config) => {
|
|
|
751
760
|
}
|
|
752
761
|
}
|
|
753
762
|
}
|
|
763
|
+
else if ((0, xmlUtils_js_1.isElement)(node) && (node.nodeName === 'm:oMath' || node.nodeName === 'oMath'
|
|
764
|
+
|| node.nodeName === 'm:oMathPara' || node.nodeName === 'oMathPara')) {
|
|
765
|
+
// Equations. Without this branch they reach the generic fallback below, which
|
|
766
|
+
// recurses into every child and concatenates the `m:t` runs with no separators -
|
|
767
|
+
// so `<m:num>1</m:num><m:den>2</m:den>` came out as "12". That is worse than
|
|
768
|
+
// dropping the formula: the result still reads as a number, so nothing downstream
|
|
769
|
+
// can tell it is wrong.
|
|
770
|
+
//
|
|
771
|
+
// `m:oMathPara` is a display equation on its own line; a bare `m:oMath` is inline.
|
|
772
|
+
const isBlock = node.nodeName === 'm:oMathPara' || node.nodeName === 'oMathPara';
|
|
773
|
+
const latex = (0, mathUtils_js_1.ommlToLatex)(node);
|
|
774
|
+
if (!(0, mathUtils_js_1.isEmptyMath)(latex)) {
|
|
775
|
+
text += latex;
|
|
776
|
+
children.push({
|
|
777
|
+
type: 'code',
|
|
778
|
+
text: latex,
|
|
779
|
+
metadata: { math: isBlock ? 'block' : 'inline' }
|
|
780
|
+
});
|
|
781
|
+
}
|
|
782
|
+
}
|
|
754
783
|
else if (node.childNodes.length > 0) {
|
|
755
784
|
// Generic fallback for unknown elements that might contain content
|
|
756
785
|
for (const child of Array.from(node.childNodes))
|
|
@@ -1071,9 +1100,6 @@ const parseWord = async (buffer, config) => {
|
|
|
1071
1100
|
if (file.path.match(footerFileRegex))
|
|
1072
1101
|
continue;
|
|
1073
1102
|
const documentContent = file.content.toString();
|
|
1074
|
-
if (config.includeRawContent) {
|
|
1075
|
-
rawContents.push(documentContent);
|
|
1076
|
-
}
|
|
1077
1103
|
const doc = (0, xmlUtils_js_1.parseXmlString)(documentContent, { locator: config.includeRawContent });
|
|
1078
1104
|
const body = (0, xmlUtils_js_1.getFirstElementByTagName)(doc, "w:body");
|
|
1079
1105
|
if (body) {
|