officeparser 7.1.0 → 7.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. package/README.md +93 -18
  2. package/dist/OfficeGenerator.d.ts +1 -1
  3. package/dist/OfficeGenerator.js +16 -7
  4. package/dist/cli.d.ts +4 -0
  5. package/dist/cli.js +12 -3
  6. package/dist/defaults.js +11 -0
  7. package/dist/generators/BaseGenerator.d.ts +3 -3
  8. package/dist/generators/ChunkingGenerator.js +8 -1
  9. package/dist/generators/CsvGenerator.d.ts +1 -1
  10. package/dist/generators/HtmlGenerator.d.ts +2 -1
  11. package/dist/generators/HtmlGenerator.js +462 -40
  12. package/dist/generators/MarkdownGenerator.d.ts +1 -1
  13. package/dist/generators/MarkdownGenerator.js +3 -1
  14. package/dist/generators/PdfGenerator.d.ts +1 -1
  15. package/dist/generators/PdfGenerator.js +0 -6
  16. package/dist/generators/RtfGenerator.d.ts +2 -1
  17. package/dist/generators/RtfGenerator.js +43 -6
  18. package/dist/generators/TextGenerator.d.ts +1 -1
  19. package/dist/officeparser.browser.d.ts +259 -52
  20. package/dist/officeparser.browser.iife.js +380 -93
  21. package/dist/officeparser.browser.mjs +380 -93
  22. package/dist/parsers/CsvParser.js +1 -1
  23. package/dist/parsers/ExcelParser.js +63 -19
  24. package/dist/parsers/HtmlParser.js +10 -1
  25. package/dist/parsers/MarkdownParser.js +13 -10
  26. package/dist/parsers/OpenOfficeParser.js +57 -34
  27. package/dist/parsers/PdfParser.js +23 -1
  28. package/dist/parsers/PowerPointParser.js +164 -40
  29. package/dist/parsers/RtfParser.js +28 -24
  30. package/dist/parsers/WordParser.js +154 -11
  31. package/dist/sbom.cdx.json +100 -100
  32. package/dist/types.d.ts +265 -52
  33. package/dist/types.js +2 -0
  34. package/dist/utils/astUtils.d.ts +2 -2
  35. package/dist/utils/astUtils.js +2 -1
  36. package/dist/utils/configUtils.d.ts +5 -0
  37. package/dist/utils/configUtils.js +55 -1
  38. package/dist/utils/errorUtils.js +2 -1
  39. package/dist/utils/xmlUtils.d.ts +9 -0
  40. package/dist/utils/xmlUtils.js +53 -1
  41. package/package.json +1 -1
@@ -52,10 +52,17 @@ const parsePowerPoint = async (buffer, config) => {
52
52
  const chartFileRegex = /ppt\/charts\/chart\d+\.xml/;
53
53
  const corePropsFileRegex = /docProps\/core\.xml/;
54
54
  const customPropsFileRegex = /docProps\/custom\.xml/;
55
+ const appPropsFileRegex = /docProps\/app\.xml/;
56
+ const commentsFileRegex = /ppt\/comments\/comment\d+\.xml/;
57
+ const commentAuthorsRegex = /ppt\/commentAuthors\.xml/;
58
+ const slideMastersRegex = /ppt\/slideMasters\/slideMaster\d+\.xml/;
55
59
  const files = await (0, zipUtils_js_1.extractFiles)(buffer, x => !!x.match(config.ignoreNotes ? slidesRegex : allFilesRegex) ||
56
60
  !!x.match(corePropsFileRegex) ||
57
61
  !!x.match(customPropsFileRegex) ||
62
+ !!x.match(appPropsFileRegex) ||
58
63
  !!x.match(slideRelsRegex) ||
64
+ (!config.ignoreComments && (!!x.match(commentsFileRegex) || !!x.match(commentAuthorsRegex))) ||
65
+ (!config.ignoreSlideMasters && !!x.match(slideMastersRegex)) ||
59
66
  (!!config.extractAttachments && (!!x.match(mediaFileRegex) || !!x.match(chartFileRegex))));
60
67
  // Extract metadata
61
68
  const corePropsFile = files.find(f => f.path.match(corePropsFileRegex));
@@ -66,6 +73,12 @@ const parsePowerPoint = async (buffer, config) => {
66
73
  if (Object.keys(customProperties).length > 0)
67
74
  metadata.customProperties = customProperties;
68
75
  }
76
+ const appPropsFile = files.find(f => f.path.match(appPropsFileRegex));
77
+ if (appPropsFile) {
78
+ const appProperties = (0, xmlUtils_js_1.parseOOXMLAppProperties)(appPropsFile.content.toString());
79
+ if (Object.keys(appProperties).length > 0)
80
+ metadata.nativeProperties = appProperties;
81
+ }
69
82
  // Sort files
70
83
  files.sort((a, b) => {
71
84
  const aMatch = a.path.match(slideNumberRegex);
@@ -77,6 +90,23 @@ const parsePowerPoint = async (buffer, config) => {
77
90
  const content = [];
78
91
  const rawContents = [];
79
92
  const slideRelsMap = {};
93
+ const authorMap = {};
94
+ if (!config.ignoreComments) {
95
+ const authorsFile = files.find(f => f.path === 'ppt/commentAuthors.xml');
96
+ if (authorsFile) {
97
+ const authorsXml = (0, xmlUtils_js_1.parseXmlString)(authorsFile.content.toString());
98
+ const authorNodes = (0, xmlUtils_js_1.getElementsByTagName)(authorsXml, "p:cmAuthor");
99
+ for (const aNode of authorNodes) {
100
+ const id = aNode.getAttribute("id");
101
+ if (id !== null) {
102
+ authorMap[id] = {
103
+ author: aNode.getAttribute("name") || undefined,
104
+ initials: aNode.getAttribute("initials") || undefined
105
+ };
106
+ }
107
+ }
108
+ }
109
+ }
80
110
  let currentListId = 0;
81
111
  let runningListIndex = 0;
82
112
  let lastWasList = false;
@@ -159,11 +189,26 @@ const parsePowerPoint = async (buffer, config) => {
159
189
  cellText += pNode.text;
160
190
  }
161
191
  }
192
+ let backgroundColor;
193
+ const tcPr = (0, xmlUtils_js_1.getFirstElementByTagName)(tcNode, "a:tcPr");
194
+ if (tcPr) {
195
+ for (const child of Array.from(tcPr.childNodes)) {
196
+ if ((0, xmlUtils_js_1.isElement)(child) && child.nodeName === "a:solidFill") {
197
+ const srgbClr = (0, xmlUtils_js_1.getFirstElementByTagName)(child, "a:srgbClr");
198
+ if (srgbClr) {
199
+ const val = srgbClr.getAttribute("val");
200
+ if (val)
201
+ backgroundColor = "#" + val;
202
+ }
203
+ break;
204
+ }
205
+ }
206
+ }
162
207
  const cellNode = {
163
208
  type: 'cell',
164
209
  text: cellText,
165
210
  children: cellChildren,
166
- metadata: { row: rIndex, col: cIndex }
211
+ metadata: { row: rIndex, col: cIndex, ...(backgroundColor ? { backgroundColor } : {}) }
167
212
  };
168
213
  cells.push(cellNode);
169
214
  }
@@ -284,12 +329,23 @@ const parsePowerPoint = async (buffer, config) => {
284
329
  const paragraphs = (0, xmlUtils_js_1.getElementsByTagName)(txBody, "a:p");
285
330
  for (let i = 0; i < paragraphs.length; i++) {
286
331
  const p = paragraphs[i];
287
- const pNode = {
288
- type: isTitle ? 'heading' : 'paragraph',
289
- text: '',
290
- children: [],
291
- metadata: isTitle ? { level: 1 } : {}
292
- };
332
+ let pNode;
333
+ if (isTitle) {
334
+ pNode = {
335
+ type: 'heading',
336
+ text: '',
337
+ children: [],
338
+ metadata: { level: 1 }
339
+ };
340
+ }
341
+ else {
342
+ pNode = {
343
+ type: 'paragraph',
344
+ text: '',
345
+ children: [],
346
+ metadata: {}
347
+ };
348
+ }
293
349
  // Paragraph Alignment and List Detection
294
350
  const pPr = (0, xmlUtils_js_1.getFirstElementByTagName)(p, "a:pPr");
295
351
  let isList = false;
@@ -330,7 +386,6 @@ const parsePowerPoint = async (buffer, config) => {
330
386
  }
331
387
  }
332
388
  if (isList) {
333
- pNode.type = 'list';
334
389
  const ilvl = lvl;
335
390
  // detect a new list when bullet type changes or previous was not a list
336
391
  const newList = !lastWasList ||
@@ -378,13 +433,17 @@ const parsePowerPoint = async (buffer, config) => {
378
433
  lastListType = listType;
379
434
  lastListIndent = ilvl;
380
435
  // metadata output
381
- pNode.metadata = {
382
- ...pNode.metadata,
383
- listType,
384
- indentation: ilvl,
385
- listId: currentListId.toString(),
386
- itemIndex: runningListIndex,
387
- alignment: pNode.metadata?.alignment || 'left',
436
+ pNode = {
437
+ type: 'list',
438
+ text: pNode.text,
439
+ children: pNode.children,
440
+ metadata: {
441
+ listType,
442
+ indentation: ilvl,
443
+ listId: currentListId.toString(),
444
+ itemIndex: runningListIndex,
445
+ alignment: pNode.metadata?.alignment || 'left',
446
+ }
388
447
  };
389
448
  }
390
449
  else {
@@ -392,7 +451,7 @@ const parsePowerPoint = async (buffer, config) => {
392
451
  lastListType = null;
393
452
  lastListIndent = 0;
394
453
  }
395
- if (isTitle) {
454
+ if (isTitle && pNode.type === 'heading') {
396
455
  pNode.metadata = { ...pNode.metadata, level: 1 };
397
456
  }
398
457
  if (config.includeRawContent) {
@@ -502,7 +561,7 @@ const parsePowerPoint = async (buffer, config) => {
502
561
  text: '',
503
562
  children: [],
504
563
  metadata: {
505
- indentation: lvl,
564
+ paragraphIndentation: { left: lvl },
506
565
  alignment: pNode.metadata?.alignment || 'left'
507
566
  }
508
567
  };
@@ -638,6 +697,10 @@ const parsePowerPoint = async (buffer, config) => {
638
697
  else if (typeAttr.includes("relationships/notesSlide")) {
639
698
  simplifiedType = "notes";
640
699
  }
700
+ // Check comments
701
+ else if (typeAttr.includes("relationships/comments")) {
702
+ simplifiedType = "comments";
703
+ }
641
704
  // Now normalize the target only if it is a local file path.
642
705
  // Hyperlinks are external and should not be normalized.
643
706
  let normalizedTarget = targetRaw;
@@ -657,6 +720,8 @@ const parsePowerPoint = async (buffer, config) => {
657
720
  }
658
721
  }
659
722
  }
723
+ const slidesMap = {};
724
+ const slideMasters = [];
660
725
  // Now for processing all the other files - slides and notes.
661
726
  for (const file of files) {
662
727
  if (file.path.match(mediaFileRegex))
@@ -667,6 +732,8 @@ const parsePowerPoint = async (buffer, config) => {
667
732
  continue;
668
733
  if (file.path.match(corePropsFileRegex))
669
734
  continue;
735
+ if (file.path.includes("comment"))
736
+ continue;
670
737
  const xmlContentString = file.content.toString();
671
738
  const xml = (0, xmlUtils_js_1.parseXmlString)(xmlContentString, { locator: config.includeRawContent });
672
739
  if (config.includeRawContent) {
@@ -674,30 +741,91 @@ const parsePowerPoint = async (buffer, config) => {
674
741
  }
675
742
  const slideMatch = file.path.match(slideNumberRegex);
676
743
  const slideNumber = slideMatch ? parseInt(slideMatch[1]) : 0;
744
+ const masterMatch = file.path.match(/slideMaster(\d+)\.xml/);
745
+ const masterNumber = masterMatch ? parseInt(masterMatch[1]) : 0;
677
746
  const isNote = file.path.includes("notesSlide");
678
- const slideNode = {
679
- type: isNote ? 'note' : 'slide',
680
- children: [],
681
- metadata: {
682
- slideNumber: slideNumber,
683
- ...(isNote ? { noteId: `slide-note-${slideNumber}` } : {})
684
- }
685
- };
747
+ const isMaster = file.path.includes("slideMaster");
748
+ const nodeType = isNote ? 'note' : (isMaster ? 'slideMaster' : 'slide');
749
+ const nodeNumber = isMaster ? masterNumber : slideNumber;
750
+ let slideNode;
751
+ if (isNote) {
752
+ slideNode = {
753
+ type: 'note',
754
+ children: [],
755
+ metadata: {
756
+ slideNumber: nodeNumber,
757
+ noteId: `slide-note-${slideNumber}`
758
+ }
759
+ };
760
+ }
761
+ else {
762
+ slideNode = {
763
+ type: isMaster ? 'slideMaster' : 'slide',
764
+ children: [],
765
+ metadata: {
766
+ slideNumber: nodeNumber
767
+ }
768
+ };
769
+ }
686
770
  if (config.includeRawContent) {
687
771
  slideNode.rawContent = (0, xmlUtils_js_1.getRawContent)(xml, xmlContentString, config);
688
772
  }
689
- /**
690
- * Extract slide contents in correct document order by scanning p:spTree children.
691
- * This ensures p:pic, p:sp, p:graphicFrame appear in AST in the exact sequence.
692
- */
693
773
  const spTree = (0, xmlUtils_js_1.getFirstElementByTagName)(xml, "p:spTree");
694
774
  if (spTree) {
695
- slideNode.children?.push(...traverseSpTree(spTree, slideNumber, xmlContentString));
775
+ slideNode.children?.push(...traverseSpTree(spTree, nodeNumber, xmlContentString));
696
776
  }
697
777
  if (slideNode.children && slideNode.children.length > 0) {
698
- content.push(slideNode);
778
+ if (isMaster) {
779
+ slideMasters.push(slideNode);
780
+ }
781
+ else if (isNote) {
782
+ if (!slidesMap[slideNumber])
783
+ slidesMap[slideNumber] = { type: 'slide', children: [], metadata: { slideNumber } };
784
+ if (!slidesMap[slideNumber].notes)
785
+ slidesMap[slideNumber].notes = [];
786
+ slidesMap[slideNumber].notes.push(slideNode);
787
+ }
788
+ else {
789
+ if (!slidesMap[slideNumber]) {
790
+ slidesMap[slideNumber] = slideNode;
791
+ }
792
+ else {
793
+ slidesMap[slideNumber].children = slideNode.children;
794
+ slidesMap[slideNumber].rawContent = slideNode.rawContent;
795
+ }
796
+ // Process comments
797
+ if (!config.ignoreComments && slideRelsMap[slideNumber]) {
798
+ const commentRels = Object.values(slideRelsMap[slideNumber]).filter(r => r.type === "comments");
799
+ for (const rel of commentRels) {
800
+ const cFile = files.find(f => f.path.endsWith(rel.target));
801
+ if (cFile) {
802
+ const cXml = (0, xmlUtils_js_1.parseXmlString)(cFile.content.toString());
803
+ const commentNodes = (0, xmlUtils_js_1.getElementsByTagName)(cXml, "p:cm");
804
+ for (const cNode of commentNodes) {
805
+ const authorId = cNode.getAttribute("authorId");
806
+ const authorData = authorId !== null ? authorMap[authorId] : undefined;
807
+ const text = (0, xmlUtils_js_1.getElementsByTagName)(cNode, "a:t").map(t => t.textContent || '').join('');
808
+ if (text) {
809
+ if (!slidesMap[slideNumber].comments)
810
+ slidesMap[slideNumber].comments = [];
811
+ slidesMap[slideNumber].comments.push({
812
+ type: 'comment',
813
+ text,
814
+ children: [{ type: 'text', text, formatting: {} }],
815
+ metadata: authorData && authorData.author ? { author: authorData.author } : undefined
816
+ });
817
+ }
818
+ }
819
+ }
820
+ }
821
+ }
822
+ }
699
823
  }
700
824
  }
825
+ const sortedSlideNumbers = Object.keys(slidesMap).map(Number).sort((a, b) => a - b);
826
+ for (const num of sortedSlideNumbers) {
827
+ content.push(slidesMap[num]);
828
+ }
701
829
  const attachments = [];
702
830
  const mediaFiles = files.filter(f => f.path.match(/ppt\/media\/.*/));
703
831
  const chartFiles = files.filter(f => f.path.match(/ppt\/charts\/chart\d+\.xml/));
@@ -762,14 +890,7 @@ const parsePowerPoint = async (buffer, config) => {
762
890
  };
763
891
  assignAttachmentData(content);
764
892
  }
765
- // Finally, if the notes are required to be at the end of the document, move them there.
766
- if (!config.ignoreNotes && config.putNotesAtLast) {
767
- content.sort((a, b) => {
768
- const aIsNote = a.type === 'note' ? 1 : 0;
769
- const bIsNote = b.type === 'note' ? 1 : 0;
770
- return aIsNote - bIsNote;
771
- });
772
- }
893
+ // putNotesAtLast is deprecated. Notes are now structurally attached to their respective slides.
773
894
  const toTextSync = () => content.map(c => {
774
895
  // Recursive text extraction
775
896
  const getText = (node) => {
@@ -783,6 +904,9 @@ const parsePowerPoint = async (buffer, config) => {
783
904
  };
784
905
  return getText(c);
785
906
  }).filter(t => t != '').join(config.newlineDelimiter);
786
- return (0, astUtils_js_1.createAST)('pptx', metadata, content, attachments, config, toTextSync);
907
+ const auxiliaryContent = slideMasters.length > 0 ? {
908
+ slideMasters
909
+ } : undefined;
910
+ return (0, astUtils_js_1.createAST)('pptx', metadata, content, attachments, config, auxiliaryContent, toTextSync);
787
911
  };
788
912
  exports.parsePowerPoint = parsePowerPoint;
@@ -1107,11 +1107,14 @@ const parseRtf = async (buffer, config) => {
1107
1107
  }
1108
1108
  // Handle footnote: switch target to notes
1109
1109
  const previousTarget = currentTarget;
1110
+ let savedParagraphTextChunks;
1111
+ let savedParagraphChildren;
1112
+ let savedParagraphRawChunks;
1110
1113
  if (isFootnote) {
1111
1114
  if (config.ignoreNotes) {
1112
1115
  return; // Skip footnote content entirely
1113
1116
  }
1114
- flushParagraph();
1117
+ flushRun();
1115
1118
  currentFootnoteId++;
1116
1119
  // Determine note type based on \fet value
1117
1120
  let noteType = 'footnote';
@@ -1135,8 +1138,25 @@ const parseRtf = async (buffer, config) => {
1135
1138
  noteType: noteType
1136
1139
  }
1137
1140
  };
1138
- notes.push(noteNode);
1141
+ if (currentParagraphChildren.length > 0) {
1142
+ const precedingNode = currentParagraphChildren[currentParagraphChildren.length - 1];
1143
+ if (!precedingNode.notes)
1144
+ precedingNode.notes = [];
1145
+ precedingNode.notes.push(noteNode);
1146
+ }
1147
+ else {
1148
+ const emptyTextNode = { type: 'text', text: '' };
1149
+ emptyTextNode.notes = [noteNode];
1150
+ currentParagraphChildren.push(emptyTextNode);
1151
+ }
1139
1152
  currentTarget = noteNode.children;
1153
+ // Save current paragraph state so we don't mix footnote paragraphs with main text
1154
+ savedParagraphTextChunks = [...currentParagraphTextChunks];
1155
+ savedParagraphChildren = [...currentParagraphChildren];
1156
+ savedParagraphRawChunks = [...currentParagraphRawChunks];
1157
+ currentParagraphTextChunks = [];
1158
+ currentParagraphChildren = [];
1159
+ currentParagraphRawChunks = [];
1140
1160
  }
1141
1161
  // Create a new formatting context for the group
1142
1162
  const groupFormatting = { ...formatting };
@@ -1195,6 +1215,10 @@ const parseRtf = async (buffer, config) => {
1195
1215
  if (isFootnote) {
1196
1216
  flushParagraph();
1197
1217
  currentTarget = previousTarget;
1218
+ // Restore the saved paragraph state
1219
+ currentParagraphTextChunks = savedParagraphTextChunks;
1220
+ currentParagraphChildren = savedParagraphChildren;
1221
+ currentParagraphRawChunks = savedParagraphRawChunks;
1198
1222
  }
1199
1223
  // Clear link URL after processing the field group
1200
1224
  if (isHyperlinkField) {
@@ -1639,18 +1663,6 @@ const parseRtf = async (buffer, config) => {
1639
1663
  flushTable();
1640
1664
  }
1641
1665
  flushParagraph();
1642
- // Notes handling:
1643
- // - If putNotesAtLast is false, notes should be added inline during traversal
1644
- // (currently they go to 'notes' array, then we append them here - this is wrong)
1645
- // - If putNotesAtLast is true, notes are appended at the very end (see below)
1646
- //
1647
- // For now, when putNotesAtLast is false, we append notes immediately after content
1648
- // This isn't truly "inline" but it's better than at the end
1649
- // TODO: Implement true inline placement during traversal
1650
- if (!config.putNotesAtLast && notes.length > 0) {
1651
- content.push(...notes);
1652
- notes.length = 0; // Clear so they don't get appended again
1653
- }
1654
1666
  // Perform OCR if enabled
1655
1667
  if (config.ocr && config.extractAttachments) {
1656
1668
  for (const attachment of attachments) {
@@ -1712,20 +1724,12 @@ const parseRtf = async (buffer, config) => {
1712
1724
  populateNoteText(content);
1713
1725
  populateNoteText(notes);
1714
1726
  const toTextSync = () => {
1715
- let text = content.map(c => c.text).join(config.newlineDelimiter);
1716
- if (config.putNotesAtLast && notes.length > 0) {
1717
- text += config.newlineDelimiter + notes.map(c => c.text).join(config.newlineDelimiter);
1718
- }
1719
- return text;
1727
+ return content.map(c => c.text).join(config.newlineDelimiter);
1720
1728
  };
1721
1729
  const result = (0, astUtils_js_1.createAST)('rtf', {
1722
1730
  // RTF Limitation: No style map available (RTF uses inline styles)
1723
1731
  }, content, attachments, // PNG and JPEG images extracted from \\pict groups
1724
- config, toTextSync);
1725
- // If putNotesAtLast is true, append notes to the end of the content array
1726
- if (config.putNotesAtLast && notes.length > 0) {
1727
- content.push(...notes);
1728
- }
1732
+ config, undefined, toTextSync);
1729
1733
  return result;
1730
1734
  };
1731
1735
  exports.parseRtf = parseRtf;