officeparser 7.0.3 → 7.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +152 -18
- package/dist/OfficeGenerator.d.ts +1 -1
- package/dist/OfficeGenerator.js +16 -7
- package/dist/OfficeParser.js +6 -0
- package/dist/cli.d.ts +4 -0
- package/dist/cli.js +12 -3
- package/dist/defaults.js +27 -1
- package/dist/generators/BaseGenerator.d.ts +3 -3
- package/dist/generators/ChunkingGenerator.js +31 -4
- package/dist/generators/CsvGenerator.d.ts +1 -1
- package/dist/generators/HtmlGenerator.d.ts +2 -1
- package/dist/generators/HtmlGenerator.js +462 -40
- package/dist/generators/MarkdownGenerator.d.ts +1 -1
- package/dist/generators/MarkdownGenerator.js +3 -1
- package/dist/generators/PdfGenerator.d.ts +1 -1
- package/dist/generators/PdfGenerator.js +51 -10
- package/dist/generators/RtfGenerator.d.ts +2 -1
- package/dist/generators/RtfGenerator.js +43 -6
- package/dist/generators/TextGenerator.d.ts +1 -1
- package/dist/officeparser.browser.d.ts +377 -53
- package/dist/officeparser.browser.iife.js +380 -93
- package/dist/officeparser.browser.mjs +380 -93
- package/dist/parsers/CsvParser.js +6 -1
- package/dist/parsers/ExcelParser.js +69 -21
- package/dist/parsers/HtmlParser.js +15 -1
- package/dist/parsers/MarkdownParser.js +18 -10
- package/dist/parsers/OpenOfficeParser.js +61 -34
- package/dist/parsers/PdfParser.js +26 -1
- package/dist/parsers/PowerPointParser.js +168 -40
- package/dist/parsers/RtfParser.js +30 -24
- package/dist/parsers/WordParser.js +158 -11
- package/dist/sbom.cdx.json +100 -100
- package/dist/types.d.ts +383 -53
- package/dist/types.js +4 -0
- package/dist/utils/astUtils.d.ts +2 -2
- package/dist/utils/astUtils.js +2 -1
- package/dist/utils/configUtils.d.ts +5 -0
- package/dist/utils/configUtils.js +69 -2
- package/dist/utils/errorUtils.d.ts +20 -0
- package/dist/utils/errorUtils.js +39 -3
- package/dist/utils/moduleLoader.js +3 -3
- package/dist/utils/ocrUtils.js +271 -66
- package/dist/utils/xmlUtils.d.ts +17 -0
- package/dist/utils/xmlUtils.js +85 -1
- package/package.json +3 -2
|
@@ -86,13 +86,21 @@ const zipUtils_js_1 = require("../utils/zipUtils.js");
|
|
|
86
86
|
* @returns A promise resolving to the parsed AST
|
|
87
87
|
*/
|
|
88
88
|
const parseWord = async (buffer, config) => {
|
|
89
|
+
// Honour cancellation requests immediately — before opening the ZIP archive, loading XML
|
|
90
|
+
// files, or kicking off any OCR work. DOCX files can be large and the inflate + XML-parse
|
|
91
|
+
// steps are synchronous-heavy, so failing fast here avoids wasted CPU time.
|
|
92
|
+
(0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
|
|
89
93
|
const documentFileRegex = /word\/document[\d+]?.xml/;
|
|
90
94
|
const footnotesFileRegex = /word\/footnotes[\d+]?.xml/;
|
|
91
95
|
const endnotesFileRegex = /word\/endnotes[\d+]?.xml/;
|
|
96
|
+
const commentsFileRegex = /word\/comments[\d+]?.xml/;
|
|
97
|
+
const headerFileRegex = /word\/header[\d+]?.xml/;
|
|
98
|
+
const footerFileRegex = /word\/footer[\d+]?.xml/;
|
|
92
99
|
const numberingFileRegex = /word\/numbering[\d+]?.xml/;
|
|
93
100
|
const mediaFileRegex = /(word\/)?media\/.*/;
|
|
94
101
|
const corePropsFileRegex = /docProps\/core[\d+]?.xml/;
|
|
95
102
|
const customPropsFileRegex = /docProps\/custom\.xml/;
|
|
103
|
+
const appPropsFileRegex = /docProps\/app[\d+]?.xml/;
|
|
96
104
|
const relsFileRegex = /word\/_rels\/document[\d+]?.xml\.rels/;
|
|
97
105
|
const stylesFileRegex = /word\/styles[\d+]?.xml/;
|
|
98
106
|
// Helper to extract formatting from run properties XML string
|
|
@@ -232,6 +240,7 @@ const parseWord = async (buffer, config) => {
|
|
|
232
240
|
!!x.match(numberingFileRegex) ||
|
|
233
241
|
!!x.match(corePropsFileRegex) ||
|
|
234
242
|
!!x.match(customPropsFileRegex) ||
|
|
243
|
+
!!x.match(appPropsFileRegex) ||
|
|
235
244
|
!!x.match(relsFileRegex) ||
|
|
236
245
|
!!x.match(stylesFileRegex) ||
|
|
237
246
|
(!!config.extractAttachments && !!x.match(mediaFileRegex)));
|
|
@@ -244,9 +253,20 @@ const parseWord = async (buffer, config) => {
|
|
|
244
253
|
if (Object.keys(customProperties).length > 0)
|
|
245
254
|
metadata.customProperties = customProperties;
|
|
246
255
|
}
|
|
256
|
+
const appPropsFile = files.find(f => f.path.match(appPropsFileRegex));
|
|
257
|
+
if (appPropsFile) {
|
|
258
|
+
const appProperties = (0, xmlUtils_js_1.parseOOXMLAppProperties)(appPropsFile.content.toString());
|
|
259
|
+
if (Object.keys(appProperties).length > 0) {
|
|
260
|
+
metadata.nativeProperties = appProperties;
|
|
261
|
+
if (appProperties['Pages'] && typeof appProperties['Pages'] === 'number') {
|
|
262
|
+
metadata.pages = appProperties['Pages'];
|
|
263
|
+
}
|
|
264
|
+
}
|
|
265
|
+
}
|
|
247
266
|
const footnoteMap = new Map();
|
|
248
267
|
const endnoteMap = new Map();
|
|
249
|
-
const
|
|
268
|
+
const commentMap = new Map();
|
|
269
|
+
const commentMetadataMap = new Map();
|
|
250
270
|
const attachments = [];
|
|
251
271
|
const mediaFiles = files.filter(f => f.path.match(mediaFileRegex));
|
|
252
272
|
// Extract relationships
|
|
@@ -449,6 +469,8 @@ const parseWord = async (buffer, config) => {
|
|
|
449
469
|
// Extract text and children
|
|
450
470
|
let text = '';
|
|
451
471
|
const children = [];
|
|
472
|
+
const notes = [];
|
|
473
|
+
const comments = [];
|
|
452
474
|
// Traverse children of paragraph (runs, hyperlinks, etc.)
|
|
453
475
|
const processChildNode = (node) => {
|
|
454
476
|
if ((0, xmlUtils_js_1.isElement)(node) && node.nodeName === 'w:r') {
|
|
@@ -616,11 +638,14 @@ const parseWord = async (buffer, config) => {
|
|
|
616
638
|
children: noteNodes,
|
|
617
639
|
metadata: { noteType: 'footnote', noteId: id }
|
|
618
640
|
};
|
|
619
|
-
if (
|
|
620
|
-
|
|
641
|
+
if (children.length > 0) {
|
|
642
|
+
const target = children[children.length - 1];
|
|
643
|
+
if (!target.notes)
|
|
644
|
+
target.notes = [];
|
|
645
|
+
target.notes.push(noteNode);
|
|
621
646
|
}
|
|
622
647
|
else {
|
|
623
|
-
|
|
648
|
+
notes.push(noteNode);
|
|
624
649
|
}
|
|
625
650
|
}
|
|
626
651
|
}
|
|
@@ -635,11 +660,40 @@ const parseWord = async (buffer, config) => {
|
|
|
635
660
|
children: noteNodes,
|
|
636
661
|
metadata: { noteType: 'endnote', noteId: id }
|
|
637
662
|
};
|
|
638
|
-
if (
|
|
639
|
-
|
|
663
|
+
if (children.length > 0) {
|
|
664
|
+
const target = children[children.length - 1];
|
|
665
|
+
if (!target.notes)
|
|
666
|
+
target.notes = [];
|
|
667
|
+
target.notes.push(noteNode);
|
|
668
|
+
}
|
|
669
|
+
else {
|
|
670
|
+
notes.push(noteNode);
|
|
671
|
+
}
|
|
672
|
+
}
|
|
673
|
+
}
|
|
674
|
+
}
|
|
675
|
+
// Comments inside runs
|
|
676
|
+
if (!config.ignoreComments) {
|
|
677
|
+
const commentRef = (0, xmlUtils_js_1.getFirstElementByTagName)(runNode, "w:commentReference");
|
|
678
|
+
if (commentRef) {
|
|
679
|
+
const id = commentRef.getAttribute("w:id");
|
|
680
|
+
if (id && commentMap.has(id)) {
|
|
681
|
+
const commentNodes = commentMap.get(id);
|
|
682
|
+
const commentInfo = commentMetadataMap.get(id);
|
|
683
|
+
const commentNode = {
|
|
684
|
+
type: 'comment',
|
|
685
|
+
text: commentNodes.map((n) => n.text).join(' '),
|
|
686
|
+
children: commentNodes,
|
|
687
|
+
metadata: commentInfo || { commentId: id }
|
|
688
|
+
};
|
|
689
|
+
if (children.length > 0) {
|
|
690
|
+
const target = children[children.length - 1];
|
|
691
|
+
if (!target.comments)
|
|
692
|
+
target.comments = [];
|
|
693
|
+
target.comments.push(commentNode);
|
|
640
694
|
}
|
|
641
695
|
else {
|
|
642
|
-
|
|
696
|
+
comments.push(commentNode);
|
|
643
697
|
}
|
|
644
698
|
}
|
|
645
699
|
}
|
|
@@ -744,6 +798,8 @@ const parseWord = async (buffer, config) => {
|
|
|
744
798
|
type: 'list',
|
|
745
799
|
text: text,
|
|
746
800
|
children: children,
|
|
801
|
+
...(notes.length > 0 ? { notes } : {}),
|
|
802
|
+
...(comments.length > 0 ? { comments } : {}),
|
|
747
803
|
metadata: {
|
|
748
804
|
listType,
|
|
749
805
|
indentation: ilvl,
|
|
@@ -765,6 +821,8 @@ const parseWord = async (buffer, config) => {
|
|
|
765
821
|
type: 'heading',
|
|
766
822
|
text: text,
|
|
767
823
|
children: children,
|
|
824
|
+
...(notes.length > 0 ? { notes } : {}),
|
|
825
|
+
...(comments.length > 0 ? { comments } : {}),
|
|
768
826
|
metadata: { level, alignment, paragraphIndentation: paraIndentation, style: pStyleVal ?? undefined, ...commonMetadata }
|
|
769
827
|
};
|
|
770
828
|
if (config.includeRawContent)
|
|
@@ -776,6 +834,8 @@ const parseWord = async (buffer, config) => {
|
|
|
776
834
|
type: 'paragraph',
|
|
777
835
|
text: text,
|
|
778
836
|
children: children,
|
|
837
|
+
...(notes.length > 0 ? { notes } : {}),
|
|
838
|
+
...(comments.length > 0 ? { comments } : {}),
|
|
779
839
|
metadata: { alignment, paragraphIndentation: paraIndentation, style: pStyleVal ?? undefined, ...commonMetadata }
|
|
780
840
|
};
|
|
781
841
|
if (config.includeRawContent)
|
|
@@ -842,6 +902,15 @@ const parseWord = async (buffer, config) => {
|
|
|
842
902
|
};
|
|
843
903
|
if (colSpan > 1)
|
|
844
904
|
cellNode.metadata.colSpan = colSpan;
|
|
905
|
+
if (tcPr) {
|
|
906
|
+
const shd = (0, xmlUtils_js_1.getFirstElementByTagName)(tcPr, "w:shd");
|
|
907
|
+
if (shd) {
|
|
908
|
+
const fill = shd.getAttribute("w:fill");
|
|
909
|
+
if (fill && fill !== "auto") {
|
|
910
|
+
cellNode.metadata.backgroundColor = "#" + fill;
|
|
911
|
+
}
|
|
912
|
+
}
|
|
913
|
+
}
|
|
845
914
|
if (isVMerge) {
|
|
846
915
|
if (vMergeRestart) {
|
|
847
916
|
vMergeMap.set(visualCol, { node: cellNode, span: 1 });
|
|
@@ -911,6 +980,77 @@ const parseWord = async (buffer, config) => {
|
|
|
911
980
|
}
|
|
912
981
|
}
|
|
913
982
|
}
|
|
983
|
+
// Pre-process comments
|
|
984
|
+
if (!config.ignoreComments) {
|
|
985
|
+
const commentsFile = files.find(f => f.path.match(commentsFileRegex));
|
|
986
|
+
if (commentsFile) {
|
|
987
|
+
const commentsDoc = (0, xmlUtils_js_1.parseXmlString)(commentsFile.content.toString());
|
|
988
|
+
const commentsXml = commentsFile.content.toString();
|
|
989
|
+
const commentNodes = (0, xmlUtils_js_1.getElementsByTagName)(commentsDoc, "w:comment");
|
|
990
|
+
for (const node of commentNodes) {
|
|
991
|
+
const id = node.getAttribute("w:id");
|
|
992
|
+
if (!id)
|
|
993
|
+
continue;
|
|
994
|
+
const author = node.getAttribute("w:author") || undefined;
|
|
995
|
+
const date = node.getAttribute("w:date") || undefined;
|
|
996
|
+
const initials = node.getAttribute("w:initials") || undefined;
|
|
997
|
+
commentMetadataMap.set(id, { commentId: id, author, date, initials });
|
|
998
|
+
const pNodes = (0, xmlUtils_js_1.getElementsByTagName)(node, "w:p");
|
|
999
|
+
commentMap.set(id, pNodes.map(p => parseParagraph(p, commentsXml)));
|
|
1000
|
+
}
|
|
1001
|
+
}
|
|
1002
|
+
}
|
|
1003
|
+
// Pre-process headers and footers
|
|
1004
|
+
const headers = [];
|
|
1005
|
+
const footers = [];
|
|
1006
|
+
if (!config.ignoreHeadersAndFooters) {
|
|
1007
|
+
const headerFiles = files.filter(f => f.path.match(headerFileRegex));
|
|
1008
|
+
for (const hFile of headerFiles) {
|
|
1009
|
+
const hDoc = (0, xmlUtils_js_1.parseXmlString)(hFile.content.toString());
|
|
1010
|
+
const hXml = hFile.content.toString();
|
|
1011
|
+
const hNodes = Array.from(hDoc.documentElement.childNodes).filter(xmlUtils_js_1.isElement);
|
|
1012
|
+
for (const child of hNodes) {
|
|
1013
|
+
if (child.nodeName === 'w:p')
|
|
1014
|
+
headers.push(parseParagraph(child, hXml));
|
|
1015
|
+
else if (child.nodeName === 'w:tbl')
|
|
1016
|
+
headers.push(parseTable(child, hXml));
|
|
1017
|
+
else if (child.nodeName === 'w:sdt') {
|
|
1018
|
+
const contentNode = (0, xmlUtils_js_1.getFirstElementByTagName)(child, "w:sdtContent");
|
|
1019
|
+
if (contentNode) {
|
|
1020
|
+
for (const sdtChild of Array.from(contentNode.childNodes).filter(xmlUtils_js_1.isElement)) {
|
|
1021
|
+
if (sdtChild.nodeName === 'w:p')
|
|
1022
|
+
headers.push(parseParagraph(sdtChild, hXml));
|
|
1023
|
+
else if (sdtChild.nodeName === 'w:tbl')
|
|
1024
|
+
headers.push(parseTable(sdtChild, hXml));
|
|
1025
|
+
}
|
|
1026
|
+
}
|
|
1027
|
+
}
|
|
1028
|
+
}
|
|
1029
|
+
}
|
|
1030
|
+
const footerFiles = files.filter(f => f.path.match(footerFileRegex));
|
|
1031
|
+
for (const fFile of footerFiles) {
|
|
1032
|
+
const fDoc = (0, xmlUtils_js_1.parseXmlString)(fFile.content.toString());
|
|
1033
|
+
const fXml = fFile.content.toString();
|
|
1034
|
+
const fNodes = Array.from(fDoc.documentElement.childNodes).filter(xmlUtils_js_1.isElement);
|
|
1035
|
+
for (const child of fNodes) {
|
|
1036
|
+
if (child.nodeName === 'w:p')
|
|
1037
|
+
footers.push(parseParagraph(child, fXml));
|
|
1038
|
+
else if (child.nodeName === 'w:tbl')
|
|
1039
|
+
footers.push(parseTable(child, fXml));
|
|
1040
|
+
else if (child.nodeName === 'w:sdt') {
|
|
1041
|
+
const contentNode = (0, xmlUtils_js_1.getFirstElementByTagName)(child, "w:sdtContent");
|
|
1042
|
+
if (contentNode) {
|
|
1043
|
+
for (const sdtChild of Array.from(contentNode.childNodes).filter(xmlUtils_js_1.isElement)) {
|
|
1044
|
+
if (sdtChild.nodeName === 'w:p')
|
|
1045
|
+
footers.push(parseParagraph(sdtChild, fXml));
|
|
1046
|
+
else if (sdtChild.nodeName === 'w:tbl')
|
|
1047
|
+
footers.push(parseTable(sdtChild, fXml));
|
|
1048
|
+
}
|
|
1049
|
+
}
|
|
1050
|
+
}
|
|
1051
|
+
}
|
|
1052
|
+
}
|
|
1053
|
+
}
|
|
914
1054
|
for (const file of files) {
|
|
915
1055
|
if (file.path.match(mediaFileRegex))
|
|
916
1056
|
continue;
|
|
@@ -924,6 +1064,12 @@ const parseWord = async (buffer, config) => {
|
|
|
924
1064
|
continue;
|
|
925
1065
|
if (file.path.match(endnotesFileRegex))
|
|
926
1066
|
continue;
|
|
1067
|
+
if (file.path.match(commentsFileRegex))
|
|
1068
|
+
continue;
|
|
1069
|
+
if (file.path.match(headerFileRegex))
|
|
1070
|
+
continue;
|
|
1071
|
+
if (file.path.match(footerFileRegex))
|
|
1072
|
+
continue;
|
|
927
1073
|
const documentContent = file.content.toString();
|
|
928
1074
|
if (config.includeRawContent) {
|
|
929
1075
|
rawContents.push(documentContent);
|
|
@@ -989,9 +1135,6 @@ const parseWord = async (buffer, config) => {
|
|
|
989
1135
|
assignOcr(content);
|
|
990
1136
|
}
|
|
991
1137
|
}
|
|
992
|
-
if (config.putNotesAtLast && collectedNotes.length > 0) {
|
|
993
|
-
content.push(...collectedNotes);
|
|
994
|
-
}
|
|
995
1138
|
const toTextSync = () => content.map(c => {
|
|
996
1139
|
// Recursive text extraction
|
|
997
1140
|
const getText = (node) => {
|
|
@@ -1008,6 +1151,10 @@ const parseWord = async (buffer, config) => {
|
|
|
1008
1151
|
};
|
|
1009
1152
|
return getText(c);
|
|
1010
1153
|
}).filter(t => t != '').join(config.newlineDelimiter);
|
|
1011
|
-
|
|
1154
|
+
const auxiliaryContent = (headers.length > 0 || footers.length > 0) ? {
|
|
1155
|
+
...(headers.length > 0 ? { headers } : {}),
|
|
1156
|
+
...(footers.length > 0 ? { footers } : {})
|
|
1157
|
+
} : undefined;
|
|
1158
|
+
return (0, astUtils_js_1.createAST)('docx', { ...metadata, formatting: docDefaults, styleMap: styleMap }, content, attachments, config, auxiliaryContent, toTextSync);
|
|
1012
1159
|
};
|
|
1013
1160
|
exports.parseWord = parseWord;
|