officeparser 7.1.0 → 7.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +93 -18
- package/dist/OfficeGenerator.d.ts +1 -1
- package/dist/OfficeGenerator.js +16 -7
- package/dist/cli.d.ts +4 -0
- package/dist/cli.js +12 -3
- package/dist/defaults.js +11 -0
- package/dist/generators/BaseGenerator.d.ts +3 -3
- package/dist/generators/ChunkingGenerator.js +8 -1
- package/dist/generators/CsvGenerator.d.ts +1 -1
- package/dist/generators/HtmlGenerator.d.ts +2 -1
- package/dist/generators/HtmlGenerator.js +462 -40
- package/dist/generators/MarkdownGenerator.d.ts +1 -1
- package/dist/generators/MarkdownGenerator.js +3 -1
- package/dist/generators/PdfGenerator.d.ts +1 -1
- package/dist/generators/PdfGenerator.js +0 -6
- package/dist/generators/RtfGenerator.d.ts +2 -1
- package/dist/generators/RtfGenerator.js +43 -6
- package/dist/generators/TextGenerator.d.ts +1 -1
- package/dist/officeparser.browser.d.ts +259 -52
- package/dist/officeparser.browser.iife.js +380 -93
- package/dist/officeparser.browser.mjs +380 -93
- package/dist/parsers/CsvParser.js +1 -1
- package/dist/parsers/ExcelParser.js +63 -19
- package/dist/parsers/HtmlParser.js +10 -1
- package/dist/parsers/MarkdownParser.js +13 -10
- package/dist/parsers/OpenOfficeParser.js +57 -34
- package/dist/parsers/PdfParser.js +23 -1
- package/dist/parsers/PowerPointParser.js +164 -40
- package/dist/parsers/RtfParser.js +28 -24
- package/dist/parsers/WordParser.js +154 -11
- package/dist/sbom.cdx.json +100 -100
- package/dist/types.d.ts +265 -52
- package/dist/types.js +2 -0
- package/dist/utils/astUtils.d.ts +2 -2
- package/dist/utils/astUtils.js +2 -1
- package/dist/utils/configUtils.d.ts +5 -0
- package/dist/utils/configUtils.js +55 -1
- package/dist/utils/errorUtils.js +2 -1
- package/dist/utils/xmlUtils.d.ts +9 -0
- package/dist/utils/xmlUtils.js +53 -1
- package/package.json +1 -1
|
@@ -93,10 +93,14 @@ const parseWord = async (buffer, config) => {
|
|
|
93
93
|
const documentFileRegex = /word\/document[\d+]?.xml/;
|
|
94
94
|
const footnotesFileRegex = /word\/footnotes[\d+]?.xml/;
|
|
95
95
|
const endnotesFileRegex = /word\/endnotes[\d+]?.xml/;
|
|
96
|
+
const commentsFileRegex = /word\/comments[\d+]?.xml/;
|
|
97
|
+
const headerFileRegex = /word\/header[\d+]?.xml/;
|
|
98
|
+
const footerFileRegex = /word\/footer[\d+]?.xml/;
|
|
96
99
|
const numberingFileRegex = /word\/numbering[\d+]?.xml/;
|
|
97
100
|
const mediaFileRegex = /(word\/)?media\/.*/;
|
|
98
101
|
const corePropsFileRegex = /docProps\/core[\d+]?.xml/;
|
|
99
102
|
const customPropsFileRegex = /docProps\/custom\.xml/;
|
|
103
|
+
const appPropsFileRegex = /docProps\/app[\d+]?.xml/;
|
|
100
104
|
const relsFileRegex = /word\/_rels\/document[\d+]?.xml\.rels/;
|
|
101
105
|
const stylesFileRegex = /word\/styles[\d+]?.xml/;
|
|
102
106
|
// Helper to extract formatting from run properties XML string
|
|
@@ -236,6 +240,7 @@ const parseWord = async (buffer, config) => {
|
|
|
236
240
|
!!x.match(numberingFileRegex) ||
|
|
237
241
|
!!x.match(corePropsFileRegex) ||
|
|
238
242
|
!!x.match(customPropsFileRegex) ||
|
|
243
|
+
!!x.match(appPropsFileRegex) ||
|
|
239
244
|
!!x.match(relsFileRegex) ||
|
|
240
245
|
!!x.match(stylesFileRegex) ||
|
|
241
246
|
(!!config.extractAttachments && !!x.match(mediaFileRegex)));
|
|
@@ -248,9 +253,20 @@ const parseWord = async (buffer, config) => {
|
|
|
248
253
|
if (Object.keys(customProperties).length > 0)
|
|
249
254
|
metadata.customProperties = customProperties;
|
|
250
255
|
}
|
|
256
|
+
const appPropsFile = files.find(f => f.path.match(appPropsFileRegex));
|
|
257
|
+
if (appPropsFile) {
|
|
258
|
+
const appProperties = (0, xmlUtils_js_1.parseOOXMLAppProperties)(appPropsFile.content.toString());
|
|
259
|
+
if (Object.keys(appProperties).length > 0) {
|
|
260
|
+
metadata.nativeProperties = appProperties;
|
|
261
|
+
if (appProperties['Pages'] && typeof appProperties['Pages'] === 'number') {
|
|
262
|
+
metadata.pages = appProperties['Pages'];
|
|
263
|
+
}
|
|
264
|
+
}
|
|
265
|
+
}
|
|
251
266
|
const footnoteMap = new Map();
|
|
252
267
|
const endnoteMap = new Map();
|
|
253
|
-
const
|
|
268
|
+
const commentMap = new Map();
|
|
269
|
+
const commentMetadataMap = new Map();
|
|
254
270
|
const attachments = [];
|
|
255
271
|
const mediaFiles = files.filter(f => f.path.match(mediaFileRegex));
|
|
256
272
|
// Extract relationships
|
|
@@ -453,6 +469,8 @@ const parseWord = async (buffer, config) => {
|
|
|
453
469
|
// Extract text and children
|
|
454
470
|
let text = '';
|
|
455
471
|
const children = [];
|
|
472
|
+
const notes = [];
|
|
473
|
+
const comments = [];
|
|
456
474
|
// Traverse children of paragraph (runs, hyperlinks, etc.)
|
|
457
475
|
const processChildNode = (node) => {
|
|
458
476
|
if ((0, xmlUtils_js_1.isElement)(node) && node.nodeName === 'w:r') {
|
|
@@ -620,11 +638,14 @@ const parseWord = async (buffer, config) => {
|
|
|
620
638
|
children: noteNodes,
|
|
621
639
|
metadata: { noteType: 'footnote', noteId: id }
|
|
622
640
|
};
|
|
623
|
-
if (
|
|
624
|
-
|
|
641
|
+
if (children.length > 0) {
|
|
642
|
+
const target = children[children.length - 1];
|
|
643
|
+
if (!target.notes)
|
|
644
|
+
target.notes = [];
|
|
645
|
+
target.notes.push(noteNode);
|
|
625
646
|
}
|
|
626
647
|
else {
|
|
627
|
-
|
|
648
|
+
notes.push(noteNode);
|
|
628
649
|
}
|
|
629
650
|
}
|
|
630
651
|
}
|
|
@@ -639,11 +660,40 @@ const parseWord = async (buffer, config) => {
|
|
|
639
660
|
children: noteNodes,
|
|
640
661
|
metadata: { noteType: 'endnote', noteId: id }
|
|
641
662
|
};
|
|
642
|
-
if (
|
|
643
|
-
|
|
663
|
+
if (children.length > 0) {
|
|
664
|
+
const target = children[children.length - 1];
|
|
665
|
+
if (!target.notes)
|
|
666
|
+
target.notes = [];
|
|
667
|
+
target.notes.push(noteNode);
|
|
668
|
+
}
|
|
669
|
+
else {
|
|
670
|
+
notes.push(noteNode);
|
|
671
|
+
}
|
|
672
|
+
}
|
|
673
|
+
}
|
|
674
|
+
}
|
|
675
|
+
// Comments inside runs
|
|
676
|
+
if (!config.ignoreComments) {
|
|
677
|
+
const commentRef = (0, xmlUtils_js_1.getFirstElementByTagName)(runNode, "w:commentReference");
|
|
678
|
+
if (commentRef) {
|
|
679
|
+
const id = commentRef.getAttribute("w:id");
|
|
680
|
+
if (id && commentMap.has(id)) {
|
|
681
|
+
const commentNodes = commentMap.get(id);
|
|
682
|
+
const commentInfo = commentMetadataMap.get(id);
|
|
683
|
+
const commentNode = {
|
|
684
|
+
type: 'comment',
|
|
685
|
+
text: commentNodes.map((n) => n.text).join(' '),
|
|
686
|
+
children: commentNodes,
|
|
687
|
+
metadata: commentInfo || { commentId: id }
|
|
688
|
+
};
|
|
689
|
+
if (children.length > 0) {
|
|
690
|
+
const target = children[children.length - 1];
|
|
691
|
+
if (!target.comments)
|
|
692
|
+
target.comments = [];
|
|
693
|
+
target.comments.push(commentNode);
|
|
644
694
|
}
|
|
645
695
|
else {
|
|
646
|
-
|
|
696
|
+
comments.push(commentNode);
|
|
647
697
|
}
|
|
648
698
|
}
|
|
649
699
|
}
|
|
@@ -748,6 +798,8 @@ const parseWord = async (buffer, config) => {
|
|
|
748
798
|
type: 'list',
|
|
749
799
|
text: text,
|
|
750
800
|
children: children,
|
|
801
|
+
...(notes.length > 0 ? { notes } : {}),
|
|
802
|
+
...(comments.length > 0 ? { comments } : {}),
|
|
751
803
|
metadata: {
|
|
752
804
|
listType,
|
|
753
805
|
indentation: ilvl,
|
|
@@ -769,6 +821,8 @@ const parseWord = async (buffer, config) => {
|
|
|
769
821
|
type: 'heading',
|
|
770
822
|
text: text,
|
|
771
823
|
children: children,
|
|
824
|
+
...(notes.length > 0 ? { notes } : {}),
|
|
825
|
+
...(comments.length > 0 ? { comments } : {}),
|
|
772
826
|
metadata: { level, alignment, paragraphIndentation: paraIndentation, style: pStyleVal ?? undefined, ...commonMetadata }
|
|
773
827
|
};
|
|
774
828
|
if (config.includeRawContent)
|
|
@@ -780,6 +834,8 @@ const parseWord = async (buffer, config) => {
|
|
|
780
834
|
type: 'paragraph',
|
|
781
835
|
text: text,
|
|
782
836
|
children: children,
|
|
837
|
+
...(notes.length > 0 ? { notes } : {}),
|
|
838
|
+
...(comments.length > 0 ? { comments } : {}),
|
|
783
839
|
metadata: { alignment, paragraphIndentation: paraIndentation, style: pStyleVal ?? undefined, ...commonMetadata }
|
|
784
840
|
};
|
|
785
841
|
if (config.includeRawContent)
|
|
@@ -846,6 +902,15 @@ const parseWord = async (buffer, config) => {
|
|
|
846
902
|
};
|
|
847
903
|
if (colSpan > 1)
|
|
848
904
|
cellNode.metadata.colSpan = colSpan;
|
|
905
|
+
if (tcPr) {
|
|
906
|
+
const shd = (0, xmlUtils_js_1.getFirstElementByTagName)(tcPr, "w:shd");
|
|
907
|
+
if (shd) {
|
|
908
|
+
const fill = shd.getAttribute("w:fill");
|
|
909
|
+
if (fill && fill !== "auto") {
|
|
910
|
+
cellNode.metadata.backgroundColor = "#" + fill;
|
|
911
|
+
}
|
|
912
|
+
}
|
|
913
|
+
}
|
|
849
914
|
if (isVMerge) {
|
|
850
915
|
if (vMergeRestart) {
|
|
851
916
|
vMergeMap.set(visualCol, { node: cellNode, span: 1 });
|
|
@@ -915,6 +980,77 @@ const parseWord = async (buffer, config) => {
|
|
|
915
980
|
}
|
|
916
981
|
}
|
|
917
982
|
}
|
|
983
|
+
// Pre-process comments
|
|
984
|
+
if (!config.ignoreComments) {
|
|
985
|
+
const commentsFile = files.find(f => f.path.match(commentsFileRegex));
|
|
986
|
+
if (commentsFile) {
|
|
987
|
+
const commentsDoc = (0, xmlUtils_js_1.parseXmlString)(commentsFile.content.toString());
|
|
988
|
+
const commentsXml = commentsFile.content.toString();
|
|
989
|
+
const commentNodes = (0, xmlUtils_js_1.getElementsByTagName)(commentsDoc, "w:comment");
|
|
990
|
+
for (const node of commentNodes) {
|
|
991
|
+
const id = node.getAttribute("w:id");
|
|
992
|
+
if (!id)
|
|
993
|
+
continue;
|
|
994
|
+
const author = node.getAttribute("w:author") || undefined;
|
|
995
|
+
const date = node.getAttribute("w:date") || undefined;
|
|
996
|
+
const initials = node.getAttribute("w:initials") || undefined;
|
|
997
|
+
commentMetadataMap.set(id, { commentId: id, author, date, initials });
|
|
998
|
+
const pNodes = (0, xmlUtils_js_1.getElementsByTagName)(node, "w:p");
|
|
999
|
+
commentMap.set(id, pNodes.map(p => parseParagraph(p, commentsXml)));
|
|
1000
|
+
}
|
|
1001
|
+
}
|
|
1002
|
+
}
|
|
1003
|
+
// Pre-process headers and footers
|
|
1004
|
+
const headers = [];
|
|
1005
|
+
const footers = [];
|
|
1006
|
+
if (!config.ignoreHeadersAndFooters) {
|
|
1007
|
+
const headerFiles = files.filter(f => f.path.match(headerFileRegex));
|
|
1008
|
+
for (const hFile of headerFiles) {
|
|
1009
|
+
const hDoc = (0, xmlUtils_js_1.parseXmlString)(hFile.content.toString());
|
|
1010
|
+
const hXml = hFile.content.toString();
|
|
1011
|
+
const hNodes = Array.from(hDoc.documentElement.childNodes).filter(xmlUtils_js_1.isElement);
|
|
1012
|
+
for (const child of hNodes) {
|
|
1013
|
+
if (child.nodeName === 'w:p')
|
|
1014
|
+
headers.push(parseParagraph(child, hXml));
|
|
1015
|
+
else if (child.nodeName === 'w:tbl')
|
|
1016
|
+
headers.push(parseTable(child, hXml));
|
|
1017
|
+
else if (child.nodeName === 'w:sdt') {
|
|
1018
|
+
const contentNode = (0, xmlUtils_js_1.getFirstElementByTagName)(child, "w:sdtContent");
|
|
1019
|
+
if (contentNode) {
|
|
1020
|
+
for (const sdtChild of Array.from(contentNode.childNodes).filter(xmlUtils_js_1.isElement)) {
|
|
1021
|
+
if (sdtChild.nodeName === 'w:p')
|
|
1022
|
+
headers.push(parseParagraph(sdtChild, hXml));
|
|
1023
|
+
else if (sdtChild.nodeName === 'w:tbl')
|
|
1024
|
+
headers.push(parseTable(sdtChild, hXml));
|
|
1025
|
+
}
|
|
1026
|
+
}
|
|
1027
|
+
}
|
|
1028
|
+
}
|
|
1029
|
+
}
|
|
1030
|
+
const footerFiles = files.filter(f => f.path.match(footerFileRegex));
|
|
1031
|
+
for (const fFile of footerFiles) {
|
|
1032
|
+
const fDoc = (0, xmlUtils_js_1.parseXmlString)(fFile.content.toString());
|
|
1033
|
+
const fXml = fFile.content.toString();
|
|
1034
|
+
const fNodes = Array.from(fDoc.documentElement.childNodes).filter(xmlUtils_js_1.isElement);
|
|
1035
|
+
for (const child of fNodes) {
|
|
1036
|
+
if (child.nodeName === 'w:p')
|
|
1037
|
+
footers.push(parseParagraph(child, fXml));
|
|
1038
|
+
else if (child.nodeName === 'w:tbl')
|
|
1039
|
+
footers.push(parseTable(child, fXml));
|
|
1040
|
+
else if (child.nodeName === 'w:sdt') {
|
|
1041
|
+
const contentNode = (0, xmlUtils_js_1.getFirstElementByTagName)(child, "w:sdtContent");
|
|
1042
|
+
if (contentNode) {
|
|
1043
|
+
for (const sdtChild of Array.from(contentNode.childNodes).filter(xmlUtils_js_1.isElement)) {
|
|
1044
|
+
if (sdtChild.nodeName === 'w:p')
|
|
1045
|
+
footers.push(parseParagraph(sdtChild, fXml));
|
|
1046
|
+
else if (sdtChild.nodeName === 'w:tbl')
|
|
1047
|
+
footers.push(parseTable(sdtChild, fXml));
|
|
1048
|
+
}
|
|
1049
|
+
}
|
|
1050
|
+
}
|
|
1051
|
+
}
|
|
1052
|
+
}
|
|
1053
|
+
}
|
|
918
1054
|
for (const file of files) {
|
|
919
1055
|
if (file.path.match(mediaFileRegex))
|
|
920
1056
|
continue;
|
|
@@ -928,6 +1064,12 @@ const parseWord = async (buffer, config) => {
|
|
|
928
1064
|
continue;
|
|
929
1065
|
if (file.path.match(endnotesFileRegex))
|
|
930
1066
|
continue;
|
|
1067
|
+
if (file.path.match(commentsFileRegex))
|
|
1068
|
+
continue;
|
|
1069
|
+
if (file.path.match(headerFileRegex))
|
|
1070
|
+
continue;
|
|
1071
|
+
if (file.path.match(footerFileRegex))
|
|
1072
|
+
continue;
|
|
931
1073
|
const documentContent = file.content.toString();
|
|
932
1074
|
if (config.includeRawContent) {
|
|
933
1075
|
rawContents.push(documentContent);
|
|
@@ -993,9 +1135,6 @@ const parseWord = async (buffer, config) => {
|
|
|
993
1135
|
assignOcr(content);
|
|
994
1136
|
}
|
|
995
1137
|
}
|
|
996
|
-
if (config.putNotesAtLast && collectedNotes.length > 0) {
|
|
997
|
-
content.push(...collectedNotes);
|
|
998
|
-
}
|
|
999
1138
|
const toTextSync = () => content.map(c => {
|
|
1000
1139
|
// Recursive text extraction
|
|
1001
1140
|
const getText = (node) => {
|
|
@@ -1012,6 +1151,10 @@ const parseWord = async (buffer, config) => {
|
|
|
1012
1151
|
};
|
|
1013
1152
|
return getText(c);
|
|
1014
1153
|
}).filter(t => t != '').join(config.newlineDelimiter);
|
|
1015
|
-
|
|
1154
|
+
const auxiliaryContent = (headers.length > 0 || footers.length > 0) ? {
|
|
1155
|
+
...(headers.length > 0 ? { headers } : {}),
|
|
1156
|
+
...(footers.length > 0 ? { footers } : {})
|
|
1157
|
+
} : undefined;
|
|
1158
|
+
return (0, astUtils_js_1.createAST)('docx', { ...metadata, formatting: docDefaults, styleMap: styleMap }, content, attachments, config, auxiliaryContent, toTextSync);
|
|
1016
1159
|
};
|
|
1017
1160
|
exports.parseWord = parseWord;
|