officeparser 7.0.3 → 7.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. package/README.md +152 -18
  2. package/dist/OfficeGenerator.d.ts +1 -1
  3. package/dist/OfficeGenerator.js +16 -7
  4. package/dist/OfficeParser.js +6 -0
  5. package/dist/cli.d.ts +4 -0
  6. package/dist/cli.js +12 -3
  7. package/dist/defaults.js +27 -1
  8. package/dist/generators/BaseGenerator.d.ts +3 -3
  9. package/dist/generators/ChunkingGenerator.js +31 -4
  10. package/dist/generators/CsvGenerator.d.ts +1 -1
  11. package/dist/generators/HtmlGenerator.d.ts +2 -1
  12. package/dist/generators/HtmlGenerator.js +462 -40
  13. package/dist/generators/MarkdownGenerator.d.ts +1 -1
  14. package/dist/generators/MarkdownGenerator.js +3 -1
  15. package/dist/generators/PdfGenerator.d.ts +1 -1
  16. package/dist/generators/PdfGenerator.js +51 -10
  17. package/dist/generators/RtfGenerator.d.ts +2 -1
  18. package/dist/generators/RtfGenerator.js +43 -6
  19. package/dist/generators/TextGenerator.d.ts +1 -1
  20. package/dist/officeparser.browser.d.ts +377 -53
  21. package/dist/officeparser.browser.iife.js +380 -93
  22. package/dist/officeparser.browser.mjs +380 -93
  23. package/dist/parsers/CsvParser.js +6 -1
  24. package/dist/parsers/ExcelParser.js +69 -21
  25. package/dist/parsers/HtmlParser.js +15 -1
  26. package/dist/parsers/MarkdownParser.js +18 -10
  27. package/dist/parsers/OpenOfficeParser.js +61 -34
  28. package/dist/parsers/PdfParser.js +26 -1
  29. package/dist/parsers/PowerPointParser.js +168 -40
  30. package/dist/parsers/RtfParser.js +30 -24
  31. package/dist/parsers/WordParser.js +158 -11
  32. package/dist/sbom.cdx.json +100 -100
  33. package/dist/types.d.ts +383 -53
  34. package/dist/types.js +4 -0
  35. package/dist/utils/astUtils.d.ts +2 -2
  36. package/dist/utils/astUtils.js +2 -1
  37. package/dist/utils/configUtils.d.ts +5 -0
  38. package/dist/utils/configUtils.js +69 -2
  39. package/dist/utils/errorUtils.d.ts +20 -0
  40. package/dist/utils/errorUtils.js +39 -3
  41. package/dist/utils/moduleLoader.js +3 -3
  42. package/dist/utils/ocrUtils.js +271 -66
  43. package/dist/utils/xmlUtils.d.ts +17 -0
  44. package/dist/utils/xmlUtils.js +85 -1
  45. package/package.json +3 -2
@@ -86,13 +86,21 @@ const zipUtils_js_1 = require("../utils/zipUtils.js");
86
86
  * @returns A promise resolving to the parsed AST
87
87
  */
88
88
  const parseWord = async (buffer, config) => {
89
+ // Honour cancellation requests immediately — before opening the ZIP archive, loading XML
90
+ // files, or kicking off any OCR work. DOCX files can be large and the inflate + XML-parse
91
+ // steps are synchronous-heavy, so failing fast here avoids wasted CPU time.
92
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
89
93
  const documentFileRegex = /word\/document[\d+]?.xml/;
90
94
  const footnotesFileRegex = /word\/footnotes[\d+]?.xml/;
91
95
  const endnotesFileRegex = /word\/endnotes[\d+]?.xml/;
96
+ const commentsFileRegex = /word\/comments[\d+]?.xml/;
97
+ const headerFileRegex = /word\/header[\d+]?.xml/;
98
+ const footerFileRegex = /word\/footer[\d+]?.xml/;
92
99
  const numberingFileRegex = /word\/numbering[\d+]?.xml/;
93
100
  const mediaFileRegex = /(word\/)?media\/.*/;
94
101
  const corePropsFileRegex = /docProps\/core[\d+]?.xml/;
95
102
  const customPropsFileRegex = /docProps\/custom\.xml/;
103
+ const appPropsFileRegex = /docProps\/app[\d+]?.xml/;
96
104
  const relsFileRegex = /word\/_rels\/document[\d+]?.xml\.rels/;
97
105
  const stylesFileRegex = /word\/styles[\d+]?.xml/;
98
106
  // Helper to extract formatting from run properties XML string
@@ -232,6 +240,7 @@ const parseWord = async (buffer, config) => {
232
240
  !!x.match(numberingFileRegex) ||
233
241
  !!x.match(corePropsFileRegex) ||
234
242
  !!x.match(customPropsFileRegex) ||
243
+ !!x.match(appPropsFileRegex) ||
235
244
  !!x.match(relsFileRegex) ||
236
245
  !!x.match(stylesFileRegex) ||
237
246
  (!!config.extractAttachments && !!x.match(mediaFileRegex)));
@@ -244,9 +253,20 @@ const parseWord = async (buffer, config) => {
244
253
  if (Object.keys(customProperties).length > 0)
245
254
  metadata.customProperties = customProperties;
246
255
  }
256
+ const appPropsFile = files.find(f => f.path.match(appPropsFileRegex));
257
+ if (appPropsFile) {
258
+ const appProperties = (0, xmlUtils_js_1.parseOOXMLAppProperties)(appPropsFile.content.toString());
259
+ if (Object.keys(appProperties).length > 0) {
260
+ metadata.nativeProperties = appProperties;
261
+ if (appProperties['Pages'] && typeof appProperties['Pages'] === 'number') {
262
+ metadata.pages = appProperties['Pages'];
263
+ }
264
+ }
265
+ }
247
266
  const footnoteMap = new Map();
248
267
  const endnoteMap = new Map();
249
- const collectedNotes = [];
268
+ const commentMap = new Map();
269
+ const commentMetadataMap = new Map();
250
270
  const attachments = [];
251
271
  const mediaFiles = files.filter(f => f.path.match(mediaFileRegex));
252
272
  // Extract relationships
@@ -449,6 +469,8 @@ const parseWord = async (buffer, config) => {
449
469
  // Extract text and children
450
470
  let text = '';
451
471
  const children = [];
472
+ const notes = [];
473
+ const comments = [];
452
474
  // Traverse children of paragraph (runs, hyperlinks, etc.)
453
475
  const processChildNode = (node) => {
454
476
  if ((0, xmlUtils_js_1.isElement)(node) && node.nodeName === 'w:r') {
@@ -616,11 +638,14 @@ const parseWord = async (buffer, config) => {
616
638
  children: noteNodes,
617
639
  metadata: { noteType: 'footnote', noteId: id }
618
640
  };
619
- if (config.putNotesAtLast) {
620
- collectedNotes.push(noteNode);
641
+ if (children.length > 0) {
642
+ const target = children[children.length - 1];
643
+ if (!target.notes)
644
+ target.notes = [];
645
+ target.notes.push(noteNode);
621
646
  }
622
647
  else {
623
- children.push(noteNode);
648
+ notes.push(noteNode);
624
649
  }
625
650
  }
626
651
  }
@@ -635,11 +660,40 @@ const parseWord = async (buffer, config) => {
635
660
  children: noteNodes,
636
661
  metadata: { noteType: 'endnote', noteId: id }
637
662
  };
638
- if (config.putNotesAtLast) {
639
- collectedNotes.push(noteNode);
663
+ if (children.length > 0) {
664
+ const target = children[children.length - 1];
665
+ if (!target.notes)
666
+ target.notes = [];
667
+ target.notes.push(noteNode);
668
+ }
669
+ else {
670
+ notes.push(noteNode);
671
+ }
672
+ }
673
+ }
674
+ }
675
+ // Comments inside runs
676
+ if (!config.ignoreComments) {
677
+ const commentRef = (0, xmlUtils_js_1.getFirstElementByTagName)(runNode, "w:commentReference");
678
+ if (commentRef) {
679
+ const id = commentRef.getAttribute("w:id");
680
+ if (id && commentMap.has(id)) {
681
+ const commentNodes = commentMap.get(id);
682
+ const commentInfo = commentMetadataMap.get(id);
683
+ const commentNode = {
684
+ type: 'comment',
685
+ text: commentNodes.map((n) => n.text).join(' '),
686
+ children: commentNodes,
687
+ metadata: commentInfo || { commentId: id }
688
+ };
689
+ if (children.length > 0) {
690
+ const target = children[children.length - 1];
691
+ if (!target.comments)
692
+ target.comments = [];
693
+ target.comments.push(commentNode);
640
694
  }
641
695
  else {
642
- children.push(noteNode);
696
+ comments.push(commentNode);
643
697
  }
644
698
  }
645
699
  }
@@ -744,6 +798,8 @@ const parseWord = async (buffer, config) => {
744
798
  type: 'list',
745
799
  text: text,
746
800
  children: children,
801
+ ...(notes.length > 0 ? { notes } : {}),
802
+ ...(comments.length > 0 ? { comments } : {}),
747
803
  metadata: {
748
804
  listType,
749
805
  indentation: ilvl,
@@ -765,6 +821,8 @@ const parseWord = async (buffer, config) => {
765
821
  type: 'heading',
766
822
  text: text,
767
823
  children: children,
824
+ ...(notes.length > 0 ? { notes } : {}),
825
+ ...(comments.length > 0 ? { comments } : {}),
768
826
  metadata: { level, alignment, paragraphIndentation: paraIndentation, style: pStyleVal ?? undefined, ...commonMetadata }
769
827
  };
770
828
  if (config.includeRawContent)
@@ -776,6 +834,8 @@ const parseWord = async (buffer, config) => {
776
834
  type: 'paragraph',
777
835
  text: text,
778
836
  children: children,
837
+ ...(notes.length > 0 ? { notes } : {}),
838
+ ...(comments.length > 0 ? { comments } : {}),
779
839
  metadata: { alignment, paragraphIndentation: paraIndentation, style: pStyleVal ?? undefined, ...commonMetadata }
780
840
  };
781
841
  if (config.includeRawContent)
@@ -842,6 +902,15 @@ const parseWord = async (buffer, config) => {
842
902
  };
843
903
  if (colSpan > 1)
844
904
  cellNode.metadata.colSpan = colSpan;
905
+ if (tcPr) {
906
+ const shd = (0, xmlUtils_js_1.getFirstElementByTagName)(tcPr, "w:shd");
907
+ if (shd) {
908
+ const fill = shd.getAttribute("w:fill");
909
+ if (fill && fill !== "auto") {
910
+ cellNode.metadata.backgroundColor = "#" + fill;
911
+ }
912
+ }
913
+ }
845
914
  if (isVMerge) {
846
915
  if (vMergeRestart) {
847
916
  vMergeMap.set(visualCol, { node: cellNode, span: 1 });
@@ -911,6 +980,77 @@ const parseWord = async (buffer, config) => {
911
980
  }
912
981
  }
913
982
  }
983
+ // Pre-process comments
984
+ if (!config.ignoreComments) {
985
+ const commentsFile = files.find(f => f.path.match(commentsFileRegex));
986
+ if (commentsFile) {
987
+ const commentsDoc = (0, xmlUtils_js_1.parseXmlString)(commentsFile.content.toString());
988
+ const commentsXml = commentsFile.content.toString();
989
+ const commentNodes = (0, xmlUtils_js_1.getElementsByTagName)(commentsDoc, "w:comment");
990
+ for (const node of commentNodes) {
991
+ const id = node.getAttribute("w:id");
992
+ if (!id)
993
+ continue;
994
+ const author = node.getAttribute("w:author") || undefined;
995
+ const date = node.getAttribute("w:date") || undefined;
996
+ const initials = node.getAttribute("w:initials") || undefined;
997
+ commentMetadataMap.set(id, { commentId: id, author, date, initials });
998
+ const pNodes = (0, xmlUtils_js_1.getElementsByTagName)(node, "w:p");
999
+ commentMap.set(id, pNodes.map(p => parseParagraph(p, commentsXml)));
1000
+ }
1001
+ }
1002
+ }
1003
+ // Pre-process headers and footers
1004
+ const headers = [];
1005
+ const footers = [];
1006
+ if (!config.ignoreHeadersAndFooters) {
1007
+ const headerFiles = files.filter(f => f.path.match(headerFileRegex));
1008
+ for (const hFile of headerFiles) {
1009
+ const hDoc = (0, xmlUtils_js_1.parseXmlString)(hFile.content.toString());
1010
+ const hXml = hFile.content.toString();
1011
+ const hNodes = Array.from(hDoc.documentElement.childNodes).filter(xmlUtils_js_1.isElement);
1012
+ for (const child of hNodes) {
1013
+ if (child.nodeName === 'w:p')
1014
+ headers.push(parseParagraph(child, hXml));
1015
+ else if (child.nodeName === 'w:tbl')
1016
+ headers.push(parseTable(child, hXml));
1017
+ else if (child.nodeName === 'w:sdt') {
1018
+ const contentNode = (0, xmlUtils_js_1.getFirstElementByTagName)(child, "w:sdtContent");
1019
+ if (contentNode) {
1020
+ for (const sdtChild of Array.from(contentNode.childNodes).filter(xmlUtils_js_1.isElement)) {
1021
+ if (sdtChild.nodeName === 'w:p')
1022
+ headers.push(parseParagraph(sdtChild, hXml));
1023
+ else if (sdtChild.nodeName === 'w:tbl')
1024
+ headers.push(parseTable(sdtChild, hXml));
1025
+ }
1026
+ }
1027
+ }
1028
+ }
1029
+ }
1030
+ const footerFiles = files.filter(f => f.path.match(footerFileRegex));
1031
+ for (const fFile of footerFiles) {
1032
+ const fDoc = (0, xmlUtils_js_1.parseXmlString)(fFile.content.toString());
1033
+ const fXml = fFile.content.toString();
1034
+ const fNodes = Array.from(fDoc.documentElement.childNodes).filter(xmlUtils_js_1.isElement);
1035
+ for (const child of fNodes) {
1036
+ if (child.nodeName === 'w:p')
1037
+ footers.push(parseParagraph(child, fXml));
1038
+ else if (child.nodeName === 'w:tbl')
1039
+ footers.push(parseTable(child, fXml));
1040
+ else if (child.nodeName === 'w:sdt') {
1041
+ const contentNode = (0, xmlUtils_js_1.getFirstElementByTagName)(child, "w:sdtContent");
1042
+ if (contentNode) {
1043
+ for (const sdtChild of Array.from(contentNode.childNodes).filter(xmlUtils_js_1.isElement)) {
1044
+ if (sdtChild.nodeName === 'w:p')
1045
+ footers.push(parseParagraph(sdtChild, fXml));
1046
+ else if (sdtChild.nodeName === 'w:tbl')
1047
+ footers.push(parseTable(sdtChild, fXml));
1048
+ }
1049
+ }
1050
+ }
1051
+ }
1052
+ }
1053
+ }
914
1054
  for (const file of files) {
915
1055
  if (file.path.match(mediaFileRegex))
916
1056
  continue;
@@ -924,6 +1064,12 @@ const parseWord = async (buffer, config) => {
924
1064
  continue;
925
1065
  if (file.path.match(endnotesFileRegex))
926
1066
  continue;
1067
+ if (file.path.match(commentsFileRegex))
1068
+ continue;
1069
+ if (file.path.match(headerFileRegex))
1070
+ continue;
1071
+ if (file.path.match(footerFileRegex))
1072
+ continue;
927
1073
  const documentContent = file.content.toString();
928
1074
  if (config.includeRawContent) {
929
1075
  rawContents.push(documentContent);
@@ -989,9 +1135,6 @@ const parseWord = async (buffer, config) => {
989
1135
  assignOcr(content);
990
1136
  }
991
1137
  }
992
- if (config.putNotesAtLast && collectedNotes.length > 0) {
993
- content.push(...collectedNotes);
994
- }
995
1138
  const toTextSync = () => content.map(c => {
996
1139
  // Recursive text extraction
997
1140
  const getText = (node) => {
@@ -1008,6 +1151,10 @@ const parseWord = async (buffer, config) => {
1008
1151
  };
1009
1152
  return getText(c);
1010
1153
  }).filter(t => t != '').join(config.newlineDelimiter);
1011
- return (0, astUtils_js_1.createAST)('docx', { ...metadata, formatting: docDefaults, styleMap: styleMap }, content, attachments, config, toTextSync);
1154
+ const auxiliaryContent = (headers.length > 0 || footers.length > 0) ? {
1155
+ ...(headers.length > 0 ? { headers } : {}),
1156
+ ...(footers.length > 0 ? { footers } : {})
1157
+ } : undefined;
1158
+ return (0, astUtils_js_1.createAST)('docx', { ...metadata, formatting: docDefaults, styleMap: styleMap }, content, attachments, config, auxiliaryContent, toTextSync);
1012
1159
  };
1013
1160
  exports.parseWord = parseWord;