officeparser 7.1.0 → 7.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. package/README.md +152 -56
  2. package/dist/OfficeGenerator.d.ts +6 -2
  3. package/dist/OfficeGenerator.js +30 -9
  4. package/dist/OfficeParser.d.ts +1 -1
  5. package/dist/OfficeParser.js +1 -1
  6. package/dist/cli.d.ts +18 -12
  7. package/dist/cli.js +255 -81
  8. package/dist/defaults.js +12 -1
  9. package/dist/generators/BaseGenerator.d.ts +4 -3
  10. package/dist/generators/BaseGenerator.js +13 -1
  11. package/dist/generators/ChunkingGenerator.js +32 -5
  12. package/dist/generators/CsvGenerator.d.ts +1 -1
  13. package/dist/generators/HtmlGenerator.d.ts +2 -1
  14. package/dist/generators/HtmlGenerator.js +481 -42
  15. package/dist/generators/MarkdownGenerator.d.ts +1 -1
  16. package/dist/generators/MarkdownGenerator.js +35 -2
  17. package/dist/generators/PdfGenerator.d.ts +1 -1
  18. package/dist/generators/PdfGenerator.js +0 -6
  19. package/dist/generators/RtfGenerator.d.ts +2 -1
  20. package/dist/generators/RtfGenerator.js +49 -6
  21. package/dist/generators/TextGenerator.d.ts +1 -1
  22. package/dist/generators/TextGenerator.js +6 -0
  23. package/dist/officeparser.browser.d.ts +267 -54
  24. package/dist/officeparser.browser.iife.js +599 -187
  25. package/dist/officeparser.browser.mjs +599 -187
  26. package/dist/parsers/CsvParser.js +1 -1
  27. package/dist/parsers/ExcelParser.js +63 -19
  28. package/dist/parsers/HtmlParser.js +10 -1
  29. package/dist/parsers/MarkdownParser.js +13 -10
  30. package/dist/parsers/OpenOfficeParser.js +57 -34
  31. package/dist/parsers/PdfParser.js +28 -3
  32. package/dist/parsers/PowerPointParser.js +164 -40
  33. package/dist/parsers/RtfParser.js +28 -24
  34. package/dist/parsers/WordParser.js +154 -11
  35. package/dist/sbom.cdx.json +100 -100
  36. package/dist/types.d.ts +268 -53
  37. package/dist/types.js +4 -0
  38. package/dist/utils/astUtils.d.ts +2 -2
  39. package/dist/utils/astUtils.js +2 -1
  40. package/dist/utils/configUtils.d.ts +5 -0
  41. package/dist/utils/configUtils.js +55 -1
  42. package/dist/utils/errorUtils.js +3 -1
  43. package/dist/utils/moduleLoader.js +55 -11
  44. package/dist/utils/xmlUtils.d.ts +9 -0
  45. package/dist/utils/xmlUtils.js +53 -1
  46. package/package.json +6 -3
@@ -93,10 +93,14 @@ const parseWord = async (buffer, config) => {
93
93
  const documentFileRegex = /word\/document[\d+]?.xml/;
94
94
  const footnotesFileRegex = /word\/footnotes[\d+]?.xml/;
95
95
  const endnotesFileRegex = /word\/endnotes[\d+]?.xml/;
96
+ const commentsFileRegex = /word\/comments[\d+]?.xml/;
97
+ const headerFileRegex = /word\/header[\d+]?.xml/;
98
+ const footerFileRegex = /word\/footer[\d+]?.xml/;
96
99
  const numberingFileRegex = /word\/numbering[\d+]?.xml/;
97
100
  const mediaFileRegex = /(word\/)?media\/.*/;
98
101
  const corePropsFileRegex = /docProps\/core[\d+]?.xml/;
99
102
  const customPropsFileRegex = /docProps\/custom\.xml/;
103
+ const appPropsFileRegex = /docProps\/app[\d+]?.xml/;
100
104
  const relsFileRegex = /word\/_rels\/document[\d+]?.xml\.rels/;
101
105
  const stylesFileRegex = /word\/styles[\d+]?.xml/;
102
106
  // Helper to extract formatting from run properties XML string
@@ -236,6 +240,7 @@ const parseWord = async (buffer, config) => {
236
240
  !!x.match(numberingFileRegex) ||
237
241
  !!x.match(corePropsFileRegex) ||
238
242
  !!x.match(customPropsFileRegex) ||
243
+ !!x.match(appPropsFileRegex) ||
239
244
  !!x.match(relsFileRegex) ||
240
245
  !!x.match(stylesFileRegex) ||
241
246
  (!!config.extractAttachments && !!x.match(mediaFileRegex)));
@@ -248,9 +253,20 @@ const parseWord = async (buffer, config) => {
248
253
  if (Object.keys(customProperties).length > 0)
249
254
  metadata.customProperties = customProperties;
250
255
  }
256
+ const appPropsFile = files.find(f => f.path.match(appPropsFileRegex));
257
+ if (appPropsFile) {
258
+ const appProperties = (0, xmlUtils_js_1.parseOOXMLAppProperties)(appPropsFile.content.toString());
259
+ if (Object.keys(appProperties).length > 0) {
260
+ metadata.nativeProperties = appProperties;
261
+ if (appProperties['Pages'] && typeof appProperties['Pages'] === 'number') {
262
+ metadata.pages = appProperties['Pages'];
263
+ }
264
+ }
265
+ }
251
266
  const footnoteMap = new Map();
252
267
  const endnoteMap = new Map();
253
- const collectedNotes = [];
268
+ const commentMap = new Map();
269
+ const commentMetadataMap = new Map();
254
270
  const attachments = [];
255
271
  const mediaFiles = files.filter(f => f.path.match(mediaFileRegex));
256
272
  // Extract relationships
@@ -453,6 +469,8 @@ const parseWord = async (buffer, config) => {
453
469
  // Extract text and children
454
470
  let text = '';
455
471
  const children = [];
472
+ const notes = [];
473
+ const comments = [];
456
474
  // Traverse children of paragraph (runs, hyperlinks, etc.)
457
475
  const processChildNode = (node) => {
458
476
  if ((0, xmlUtils_js_1.isElement)(node) && node.nodeName === 'w:r') {
@@ -620,11 +638,14 @@ const parseWord = async (buffer, config) => {
620
638
  children: noteNodes,
621
639
  metadata: { noteType: 'footnote', noteId: id }
622
640
  };
623
- if (config.putNotesAtLast) {
624
- collectedNotes.push(noteNode);
641
+ if (children.length > 0) {
642
+ const target = children[children.length - 1];
643
+ if (!target.notes)
644
+ target.notes = [];
645
+ target.notes.push(noteNode);
625
646
  }
626
647
  else {
627
- children.push(noteNode);
648
+ notes.push(noteNode);
628
649
  }
629
650
  }
630
651
  }
@@ -639,11 +660,40 @@ const parseWord = async (buffer, config) => {
639
660
  children: noteNodes,
640
661
  metadata: { noteType: 'endnote', noteId: id }
641
662
  };
642
- if (config.putNotesAtLast) {
643
- collectedNotes.push(noteNode);
663
+ if (children.length > 0) {
664
+ const target = children[children.length - 1];
665
+ if (!target.notes)
666
+ target.notes = [];
667
+ target.notes.push(noteNode);
668
+ }
669
+ else {
670
+ notes.push(noteNode);
671
+ }
672
+ }
673
+ }
674
+ }
675
+ // Comments inside runs
676
+ if (!config.ignoreComments) {
677
+ const commentRef = (0, xmlUtils_js_1.getFirstElementByTagName)(runNode, "w:commentReference");
678
+ if (commentRef) {
679
+ const id = commentRef.getAttribute("w:id");
680
+ if (id && commentMap.has(id)) {
681
+ const commentNodes = commentMap.get(id);
682
+ const commentInfo = commentMetadataMap.get(id);
683
+ const commentNode = {
684
+ type: 'comment',
685
+ text: commentNodes.map((n) => n.text).join(' '),
686
+ children: commentNodes,
687
+ metadata: commentInfo || { commentId: id }
688
+ };
689
+ if (children.length > 0) {
690
+ const target = children[children.length - 1];
691
+ if (!target.comments)
692
+ target.comments = [];
693
+ target.comments.push(commentNode);
644
694
  }
645
695
  else {
646
- children.push(noteNode);
696
+ comments.push(commentNode);
647
697
  }
648
698
  }
649
699
  }
@@ -748,6 +798,8 @@ const parseWord = async (buffer, config) => {
748
798
  type: 'list',
749
799
  text: text,
750
800
  children: children,
801
+ ...(notes.length > 0 ? { notes } : {}),
802
+ ...(comments.length > 0 ? { comments } : {}),
751
803
  metadata: {
752
804
  listType,
753
805
  indentation: ilvl,
@@ -769,6 +821,8 @@ const parseWord = async (buffer, config) => {
769
821
  type: 'heading',
770
822
  text: text,
771
823
  children: children,
824
+ ...(notes.length > 0 ? { notes } : {}),
825
+ ...(comments.length > 0 ? { comments } : {}),
772
826
  metadata: { level, alignment, paragraphIndentation: paraIndentation, style: pStyleVal ?? undefined, ...commonMetadata }
773
827
  };
774
828
  if (config.includeRawContent)
@@ -780,6 +834,8 @@ const parseWord = async (buffer, config) => {
780
834
  type: 'paragraph',
781
835
  text: text,
782
836
  children: children,
837
+ ...(notes.length > 0 ? { notes } : {}),
838
+ ...(comments.length > 0 ? { comments } : {}),
783
839
  metadata: { alignment, paragraphIndentation: paraIndentation, style: pStyleVal ?? undefined, ...commonMetadata }
784
840
  };
785
841
  if (config.includeRawContent)
@@ -846,6 +902,15 @@ const parseWord = async (buffer, config) => {
846
902
  };
847
903
  if (colSpan > 1)
848
904
  cellNode.metadata.colSpan = colSpan;
905
+ if (tcPr) {
906
+ const shd = (0, xmlUtils_js_1.getFirstElementByTagName)(tcPr, "w:shd");
907
+ if (shd) {
908
+ const fill = shd.getAttribute("w:fill");
909
+ if (fill && fill !== "auto") {
910
+ cellNode.metadata.backgroundColor = "#" + fill;
911
+ }
912
+ }
913
+ }
849
914
  if (isVMerge) {
850
915
  if (vMergeRestart) {
851
916
  vMergeMap.set(visualCol, { node: cellNode, span: 1 });
@@ -915,6 +980,77 @@ const parseWord = async (buffer, config) => {
915
980
  }
916
981
  }
917
982
  }
983
+ // Pre-process comments
984
+ if (!config.ignoreComments) {
985
+ const commentsFile = files.find(f => f.path.match(commentsFileRegex));
986
+ if (commentsFile) {
987
+ const commentsDoc = (0, xmlUtils_js_1.parseXmlString)(commentsFile.content.toString());
988
+ const commentsXml = commentsFile.content.toString();
989
+ const commentNodes = (0, xmlUtils_js_1.getElementsByTagName)(commentsDoc, "w:comment");
990
+ for (const node of commentNodes) {
991
+ const id = node.getAttribute("w:id");
992
+ if (!id)
993
+ continue;
994
+ const author = node.getAttribute("w:author") || undefined;
995
+ const date = node.getAttribute("w:date") || undefined;
996
+ const initials = node.getAttribute("w:initials") || undefined;
997
+ commentMetadataMap.set(id, { commentId: id, author, date, initials });
998
+ const pNodes = (0, xmlUtils_js_1.getElementsByTagName)(node, "w:p");
999
+ commentMap.set(id, pNodes.map(p => parseParagraph(p, commentsXml)));
1000
+ }
1001
+ }
1002
+ }
1003
+ // Pre-process headers and footers
1004
+ const headers = [];
1005
+ const footers = [];
1006
+ if (!config.ignoreHeadersAndFooters) {
1007
+ const headerFiles = files.filter(f => f.path.match(headerFileRegex));
1008
+ for (const hFile of headerFiles) {
1009
+ const hDoc = (0, xmlUtils_js_1.parseXmlString)(hFile.content.toString());
1010
+ const hXml = hFile.content.toString();
1011
+ const hNodes = Array.from(hDoc.documentElement.childNodes).filter(xmlUtils_js_1.isElement);
1012
+ for (const child of hNodes) {
1013
+ if (child.nodeName === 'w:p')
1014
+ headers.push(parseParagraph(child, hXml));
1015
+ else if (child.nodeName === 'w:tbl')
1016
+ headers.push(parseTable(child, hXml));
1017
+ else if (child.nodeName === 'w:sdt') {
1018
+ const contentNode = (0, xmlUtils_js_1.getFirstElementByTagName)(child, "w:sdtContent");
1019
+ if (contentNode) {
1020
+ for (const sdtChild of Array.from(contentNode.childNodes).filter(xmlUtils_js_1.isElement)) {
1021
+ if (sdtChild.nodeName === 'w:p')
1022
+ headers.push(parseParagraph(sdtChild, hXml));
1023
+ else if (sdtChild.nodeName === 'w:tbl')
1024
+ headers.push(parseTable(sdtChild, hXml));
1025
+ }
1026
+ }
1027
+ }
1028
+ }
1029
+ }
1030
+ const footerFiles = files.filter(f => f.path.match(footerFileRegex));
1031
+ for (const fFile of footerFiles) {
1032
+ const fDoc = (0, xmlUtils_js_1.parseXmlString)(fFile.content.toString());
1033
+ const fXml = fFile.content.toString();
1034
+ const fNodes = Array.from(fDoc.documentElement.childNodes).filter(xmlUtils_js_1.isElement);
1035
+ for (const child of fNodes) {
1036
+ if (child.nodeName === 'w:p')
1037
+ footers.push(parseParagraph(child, fXml));
1038
+ else if (child.nodeName === 'w:tbl')
1039
+ footers.push(parseTable(child, fXml));
1040
+ else if (child.nodeName === 'w:sdt') {
1041
+ const contentNode = (0, xmlUtils_js_1.getFirstElementByTagName)(child, "w:sdtContent");
1042
+ if (contentNode) {
1043
+ for (const sdtChild of Array.from(contentNode.childNodes).filter(xmlUtils_js_1.isElement)) {
1044
+ if (sdtChild.nodeName === 'w:p')
1045
+ footers.push(parseParagraph(sdtChild, fXml));
1046
+ else if (sdtChild.nodeName === 'w:tbl')
1047
+ footers.push(parseTable(sdtChild, fXml));
1048
+ }
1049
+ }
1050
+ }
1051
+ }
1052
+ }
1053
+ }
918
1054
  for (const file of files) {
919
1055
  if (file.path.match(mediaFileRegex))
920
1056
  continue;
@@ -928,6 +1064,12 @@ const parseWord = async (buffer, config) => {
928
1064
  continue;
929
1065
  if (file.path.match(endnotesFileRegex))
930
1066
  continue;
1067
+ if (file.path.match(commentsFileRegex))
1068
+ continue;
1069
+ if (file.path.match(headerFileRegex))
1070
+ continue;
1071
+ if (file.path.match(footerFileRegex))
1072
+ continue;
931
1073
  const documentContent = file.content.toString();
932
1074
  if (config.includeRawContent) {
933
1075
  rawContents.push(documentContent);
@@ -993,9 +1135,6 @@ const parseWord = async (buffer, config) => {
993
1135
  assignOcr(content);
994
1136
  }
995
1137
  }
996
- if (config.putNotesAtLast && collectedNotes.length > 0) {
997
- content.push(...collectedNotes);
998
- }
999
1138
  const toTextSync = () => content.map(c => {
1000
1139
  // Recursive text extraction
1001
1140
  const getText = (node) => {
@@ -1012,6 +1151,10 @@ const parseWord = async (buffer, config) => {
1012
1151
  };
1013
1152
  return getText(c);
1014
1153
  }).filter(t => t != '').join(config.newlineDelimiter);
1015
- return (0, astUtils_js_1.createAST)('docx', { ...metadata, formatting: docDefaults, styleMap: styleMap }, content, attachments, config, toTextSync);
1154
+ const auxiliaryContent = (headers.length > 0 || footers.length > 0) ? {
1155
+ ...(headers.length > 0 ? { headers } : {}),
1156
+ ...(footers.length > 0 ? { footers } : {})
1157
+ } : undefined;
1158
+ return (0, astUtils_js_1.createAST)('docx', { ...metadata, formatting: docDefaults, styleMap: styleMap }, content, attachments, config, auxiliaryContent, toTextSync);
1016
1159
  };
1017
1160
  exports.parseWord = parseWord;