officeparser 7.0.3 → 7.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. package/README.md +152 -18
  2. package/dist/OfficeGenerator.d.ts +1 -1
  3. package/dist/OfficeGenerator.js +16 -7
  4. package/dist/OfficeParser.js +6 -0
  5. package/dist/cli.d.ts +4 -0
  6. package/dist/cli.js +12 -3
  7. package/dist/defaults.js +27 -1
  8. package/dist/generators/BaseGenerator.d.ts +3 -3
  9. package/dist/generators/ChunkingGenerator.js +31 -4
  10. package/dist/generators/CsvGenerator.d.ts +1 -1
  11. package/dist/generators/HtmlGenerator.d.ts +2 -1
  12. package/dist/generators/HtmlGenerator.js +462 -40
  13. package/dist/generators/MarkdownGenerator.d.ts +1 -1
  14. package/dist/generators/MarkdownGenerator.js +3 -1
  15. package/dist/generators/PdfGenerator.d.ts +1 -1
  16. package/dist/generators/PdfGenerator.js +51 -10
  17. package/dist/generators/RtfGenerator.d.ts +2 -1
  18. package/dist/generators/RtfGenerator.js +43 -6
  19. package/dist/generators/TextGenerator.d.ts +1 -1
  20. package/dist/officeparser.browser.d.ts +377 -53
  21. package/dist/officeparser.browser.iife.js +380 -93
  22. package/dist/officeparser.browser.mjs +380 -93
  23. package/dist/parsers/CsvParser.js +6 -1
  24. package/dist/parsers/ExcelParser.js +69 -21
  25. package/dist/parsers/HtmlParser.js +15 -1
  26. package/dist/parsers/MarkdownParser.js +18 -10
  27. package/dist/parsers/OpenOfficeParser.js +61 -34
  28. package/dist/parsers/PdfParser.js +26 -1
  29. package/dist/parsers/PowerPointParser.js +168 -40
  30. package/dist/parsers/RtfParser.js +30 -24
  31. package/dist/parsers/WordParser.js +158 -11
  32. package/dist/sbom.cdx.json +100 -100
  33. package/dist/types.d.ts +383 -53
  34. package/dist/types.js +4 -0
  35. package/dist/utils/astUtils.d.ts +2 -2
  36. package/dist/utils/astUtils.js +2 -1
  37. package/dist/utils/configUtils.d.ts +5 -0
  38. package/dist/utils/configUtils.js +69 -2
  39. package/dist/utils/errorUtils.d.ts +20 -0
  40. package/dist/utils/errorUtils.js +39 -3
  41. package/dist/utils/moduleLoader.js +3 -3
  42. package/dist/utils/ocrUtils.js +271 -66
  43. package/dist/utils/xmlUtils.d.ts +17 -0
  44. package/dist/utils/xmlUtils.js +85 -1
  45. package/package.json +3 -2
@@ -40,6 +40,10 @@ const zipUtils_js_1 = require("../utils/zipUtils.js");
40
40
  * @returns A promise resolving to the parsed AST
41
41
  */
42
42
  const parsePowerPoint = async (buffer, config) => {
43
+ // Honour cancellation requests immediately — before extracting the ZIP archive.
44
+ // PPTX presentations can have many slides with media/charts and optional OCR per image,
45
+ // so an early abort prevents decompressing and traversing data that will be discarded.
46
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
43
47
  const allFilesRegex = /ppt\/(notesSlides|slides)\/(notesSlide|slide)\d+.xml/g;
44
48
  const slidesRegex = /ppt\/slides\/slide\d+.xml/g;
45
49
  const slideRelsRegex = /ppt\/slides\/_rels\/slide\d+\.xml\.rels/;
@@ -48,10 +52,17 @@ const parsePowerPoint = async (buffer, config) => {
48
52
  const chartFileRegex = /ppt\/charts\/chart\d+\.xml/;
49
53
  const corePropsFileRegex = /docProps\/core\.xml/;
50
54
  const customPropsFileRegex = /docProps\/custom\.xml/;
55
+ const appPropsFileRegex = /docProps\/app\.xml/;
56
+ const commentsFileRegex = /ppt\/comments\/comment\d+\.xml/;
57
+ const commentAuthorsRegex = /ppt\/commentAuthors\.xml/;
58
+ const slideMastersRegex = /ppt\/slideMasters\/slideMaster\d+\.xml/;
51
59
  const files = await (0, zipUtils_js_1.extractFiles)(buffer, x => !!x.match(config.ignoreNotes ? slidesRegex : allFilesRegex) ||
52
60
  !!x.match(corePropsFileRegex) ||
53
61
  !!x.match(customPropsFileRegex) ||
62
+ !!x.match(appPropsFileRegex) ||
54
63
  !!x.match(slideRelsRegex) ||
64
+ (!config.ignoreComments && (!!x.match(commentsFileRegex) || !!x.match(commentAuthorsRegex))) ||
65
+ (!config.ignoreSlideMasters && !!x.match(slideMastersRegex)) ||
55
66
  (!!config.extractAttachments && (!!x.match(mediaFileRegex) || !!x.match(chartFileRegex))));
56
67
  // Extract metadata
57
68
  const corePropsFile = files.find(f => f.path.match(corePropsFileRegex));
@@ -62,6 +73,12 @@ const parsePowerPoint = async (buffer, config) => {
62
73
  if (Object.keys(customProperties).length > 0)
63
74
  metadata.customProperties = customProperties;
64
75
  }
76
+ const appPropsFile = files.find(f => f.path.match(appPropsFileRegex));
77
+ if (appPropsFile) {
78
+ const appProperties = (0, xmlUtils_js_1.parseOOXMLAppProperties)(appPropsFile.content.toString());
79
+ if (Object.keys(appProperties).length > 0)
80
+ metadata.nativeProperties = appProperties;
81
+ }
65
82
  // Sort files
66
83
  files.sort((a, b) => {
67
84
  const aMatch = a.path.match(slideNumberRegex);
@@ -73,6 +90,23 @@ const parsePowerPoint = async (buffer, config) => {
73
90
  const content = [];
74
91
  const rawContents = [];
75
92
  const slideRelsMap = {};
93
+ const authorMap = {};
94
+ if (!config.ignoreComments) {
95
+ const authorsFile = files.find(f => f.path === 'ppt/commentAuthors.xml');
96
+ if (authorsFile) {
97
+ const authorsXml = (0, xmlUtils_js_1.parseXmlString)(authorsFile.content.toString());
98
+ const authorNodes = (0, xmlUtils_js_1.getElementsByTagName)(authorsXml, "p:cmAuthor");
99
+ for (const aNode of authorNodes) {
100
+ const id = aNode.getAttribute("id");
101
+ if (id !== null) {
102
+ authorMap[id] = {
103
+ author: aNode.getAttribute("name") || undefined,
104
+ initials: aNode.getAttribute("initials") || undefined
105
+ };
106
+ }
107
+ }
108
+ }
109
+ }
76
110
  let currentListId = 0;
77
111
  let runningListIndex = 0;
78
112
  let lastWasList = false;
@@ -155,11 +189,26 @@ const parsePowerPoint = async (buffer, config) => {
155
189
  cellText += pNode.text;
156
190
  }
157
191
  }
192
+ let backgroundColor;
193
+ const tcPr = (0, xmlUtils_js_1.getFirstElementByTagName)(tcNode, "a:tcPr");
194
+ if (tcPr) {
195
+ for (const child of Array.from(tcPr.childNodes)) {
196
+ if ((0, xmlUtils_js_1.isElement)(child) && child.nodeName === "a:solidFill") {
197
+ const srgbClr = (0, xmlUtils_js_1.getFirstElementByTagName)(child, "a:srgbClr");
198
+ if (srgbClr) {
199
+ const val = srgbClr.getAttribute("val");
200
+ if (val)
201
+ backgroundColor = "#" + val;
202
+ }
203
+ break;
204
+ }
205
+ }
206
+ }
158
207
  const cellNode = {
159
208
  type: 'cell',
160
209
  text: cellText,
161
210
  children: cellChildren,
162
- metadata: { row: rIndex, col: cIndex }
211
+ metadata: { row: rIndex, col: cIndex, ...(backgroundColor ? { backgroundColor } : {}) }
163
212
  };
164
213
  cells.push(cellNode);
165
214
  }
@@ -280,12 +329,23 @@ const parsePowerPoint = async (buffer, config) => {
280
329
  const paragraphs = (0, xmlUtils_js_1.getElementsByTagName)(txBody, "a:p");
281
330
  for (let i = 0; i < paragraphs.length; i++) {
282
331
  const p = paragraphs[i];
283
- const pNode = {
284
- type: isTitle ? 'heading' : 'paragraph',
285
- text: '',
286
- children: [],
287
- metadata: isTitle ? { level: 1 } : {}
288
- };
332
+ let pNode;
333
+ if (isTitle) {
334
+ pNode = {
335
+ type: 'heading',
336
+ text: '',
337
+ children: [],
338
+ metadata: { level: 1 }
339
+ };
340
+ }
341
+ else {
342
+ pNode = {
343
+ type: 'paragraph',
344
+ text: '',
345
+ children: [],
346
+ metadata: {}
347
+ };
348
+ }
289
349
  // Paragraph Alignment and List Detection
290
350
  const pPr = (0, xmlUtils_js_1.getFirstElementByTagName)(p, "a:pPr");
291
351
  let isList = false;
@@ -326,7 +386,6 @@ const parsePowerPoint = async (buffer, config) => {
326
386
  }
327
387
  }
328
388
  if (isList) {
329
- pNode.type = 'list';
330
389
  const ilvl = lvl;
331
390
  // detect a new list when bullet type changes or previous was not a list
332
391
  const newList = !lastWasList ||
@@ -374,13 +433,17 @@ const parsePowerPoint = async (buffer, config) => {
374
433
  lastListType = listType;
375
434
  lastListIndent = ilvl;
376
435
  // metadata output
377
- pNode.metadata = {
378
- ...pNode.metadata,
379
- listType,
380
- indentation: ilvl,
381
- listId: currentListId.toString(),
382
- itemIndex: runningListIndex,
383
- alignment: pNode.metadata?.alignment || 'left',
436
+ pNode = {
437
+ type: 'list',
438
+ text: pNode.text,
439
+ children: pNode.children,
440
+ metadata: {
441
+ listType,
442
+ indentation: ilvl,
443
+ listId: currentListId.toString(),
444
+ itemIndex: runningListIndex,
445
+ alignment: pNode.metadata?.alignment || 'left',
446
+ }
384
447
  };
385
448
  }
386
449
  else {
@@ -388,7 +451,7 @@ const parsePowerPoint = async (buffer, config) => {
388
451
  lastListType = null;
389
452
  lastListIndent = 0;
390
453
  }
391
- if (isTitle) {
454
+ if (isTitle && pNode.type === 'heading') {
392
455
  pNode.metadata = { ...pNode.metadata, level: 1 };
393
456
  }
394
457
  if (config.includeRawContent) {
@@ -498,7 +561,7 @@ const parsePowerPoint = async (buffer, config) => {
498
561
  text: '',
499
562
  children: [],
500
563
  metadata: {
501
- indentation: lvl,
564
+ paragraphIndentation: { left: lvl },
502
565
  alignment: pNode.metadata?.alignment || 'left'
503
566
  }
504
567
  };
@@ -634,6 +697,10 @@ const parsePowerPoint = async (buffer, config) => {
634
697
  else if (typeAttr.includes("relationships/notesSlide")) {
635
698
  simplifiedType = "notes";
636
699
  }
700
+ // Check comments
701
+ else if (typeAttr.includes("relationships/comments")) {
702
+ simplifiedType = "comments";
703
+ }
637
704
  // Now normalize the target only if it is a local file path.
638
705
  // Hyperlinks are external and should not be normalized.
639
706
  let normalizedTarget = targetRaw;
@@ -653,6 +720,8 @@ const parsePowerPoint = async (buffer, config) => {
653
720
  }
654
721
  }
655
722
  }
723
+ const slidesMap = {};
724
+ const slideMasters = [];
656
725
  // Now for processing all the other files - slides and notes.
657
726
  for (const file of files) {
658
727
  if (file.path.match(mediaFileRegex))
@@ -663,6 +732,8 @@ const parsePowerPoint = async (buffer, config) => {
663
732
  continue;
664
733
  if (file.path.match(corePropsFileRegex))
665
734
  continue;
735
+ if (file.path.includes("comment"))
736
+ continue;
666
737
  const xmlContentString = file.content.toString();
667
738
  const xml = (0, xmlUtils_js_1.parseXmlString)(xmlContentString, { locator: config.includeRawContent });
668
739
  if (config.includeRawContent) {
@@ -670,30 +741,91 @@ const parsePowerPoint = async (buffer, config) => {
670
741
  }
671
742
  const slideMatch = file.path.match(slideNumberRegex);
672
743
  const slideNumber = slideMatch ? parseInt(slideMatch[1]) : 0;
744
+ const masterMatch = file.path.match(/slideMaster(\d+)\.xml/);
745
+ const masterNumber = masterMatch ? parseInt(masterMatch[1]) : 0;
673
746
  const isNote = file.path.includes("notesSlide");
674
- const slideNode = {
675
- type: isNote ? 'note' : 'slide',
676
- children: [],
677
- metadata: {
678
- slideNumber: slideNumber,
679
- ...(isNote ? { noteId: `slide-note-${slideNumber}` } : {})
680
- }
681
- };
747
+ const isMaster = file.path.includes("slideMaster");
748
+ const nodeType = isNote ? 'note' : (isMaster ? 'slideMaster' : 'slide');
749
+ const nodeNumber = isMaster ? masterNumber : slideNumber;
750
+ let slideNode;
751
+ if (isNote) {
752
+ slideNode = {
753
+ type: 'note',
754
+ children: [],
755
+ metadata: {
756
+ slideNumber: nodeNumber,
757
+ noteId: `slide-note-${slideNumber}`
758
+ }
759
+ };
760
+ }
761
+ else {
762
+ slideNode = {
763
+ type: isMaster ? 'slideMaster' : 'slide',
764
+ children: [],
765
+ metadata: {
766
+ slideNumber: nodeNumber
767
+ }
768
+ };
769
+ }
682
770
  if (config.includeRawContent) {
683
771
  slideNode.rawContent = (0, xmlUtils_js_1.getRawContent)(xml, xmlContentString, config);
684
772
  }
685
- /**
686
- * Extract slide contents in correct document order by scanning p:spTree children.
687
- * This ensures p:pic, p:sp, p:graphicFrame appear in AST in the exact sequence.
688
- */
689
773
  const spTree = (0, xmlUtils_js_1.getFirstElementByTagName)(xml, "p:spTree");
690
774
  if (spTree) {
691
- slideNode.children?.push(...traverseSpTree(spTree, slideNumber, xmlContentString));
775
+ slideNode.children?.push(...traverseSpTree(spTree, nodeNumber, xmlContentString));
692
776
  }
693
777
  if (slideNode.children && slideNode.children.length > 0) {
694
- content.push(slideNode);
778
+ if (isMaster) {
779
+ slideMasters.push(slideNode);
780
+ }
781
+ else if (isNote) {
782
+ if (!slidesMap[slideNumber])
783
+ slidesMap[slideNumber] = { type: 'slide', children: [], metadata: { slideNumber } };
784
+ if (!slidesMap[slideNumber].notes)
785
+ slidesMap[slideNumber].notes = [];
786
+ slidesMap[slideNumber].notes.push(slideNode);
787
+ }
788
+ else {
789
+ if (!slidesMap[slideNumber]) {
790
+ slidesMap[slideNumber] = slideNode;
791
+ }
792
+ else {
793
+ slidesMap[slideNumber].children = slideNode.children;
794
+ slidesMap[slideNumber].rawContent = slideNode.rawContent;
795
+ }
796
+ // Process comments
797
+ if (!config.ignoreComments && slideRelsMap[slideNumber]) {
798
+ const commentRels = Object.values(slideRelsMap[slideNumber]).filter(r => r.type === "comments");
799
+ for (const rel of commentRels) {
800
+ const cFile = files.find(f => f.path.endsWith(rel.target));
801
+ if (cFile) {
802
+ const cXml = (0, xmlUtils_js_1.parseXmlString)(cFile.content.toString());
803
+ const commentNodes = (0, xmlUtils_js_1.getElementsByTagName)(cXml, "p:cm");
804
+ for (const cNode of commentNodes) {
805
+ const authorId = cNode.getAttribute("authorId");
806
+ const authorData = authorId !== null ? authorMap[authorId] : undefined;
807
+ const text = (0, xmlUtils_js_1.getElementsByTagName)(cNode, "a:t").map(t => t.textContent || '').join('');
808
+ if (text) {
809
+ if (!slidesMap[slideNumber].comments)
810
+ slidesMap[slideNumber].comments = [];
811
+ slidesMap[slideNumber].comments.push({
812
+ type: 'comment',
813
+ text,
814
+ children: [{ type: 'text', text, formatting: {} }],
815
+ metadata: authorData && authorData.author ? { author: authorData.author } : undefined
816
+ });
817
+ }
818
+ }
819
+ }
820
+ }
821
+ }
822
+ }
695
823
  }
696
824
  }
825
+ const sortedSlideNumbers = Object.keys(slidesMap).map(Number).sort((a, b) => a - b);
826
+ for (const num of sortedSlideNumbers) {
827
+ content.push(slidesMap[num]);
828
+ }
697
829
  const attachments = [];
698
830
  const mediaFiles = files.filter(f => f.path.match(/ppt\/media\/.*/));
699
831
  const chartFiles = files.filter(f => f.path.match(/ppt\/charts\/chart\d+\.xml/));
@@ -758,14 +890,7 @@ const parsePowerPoint = async (buffer, config) => {
758
890
  };
759
891
  assignAttachmentData(content);
760
892
  }
761
- // Finally, if the notes are required to be at the end of the document, move them there.
762
- if (!config.ignoreNotes && config.putNotesAtLast) {
763
- content.sort((a, b) => {
764
- const aIsNote = a.type === 'note' ? 1 : 0;
765
- const bIsNote = b.type === 'note' ? 1 : 0;
766
- return aIsNote - bIsNote;
767
- });
768
- }
893
+ // putNotesAtLast is deprecated. Notes are now structurally attached to their respective slides.
769
894
  const toTextSync = () => content.map(c => {
770
895
  // Recursive text extraction
771
896
  const getText = (node) => {
@@ -779,6 +904,9 @@ const parsePowerPoint = async (buffer, config) => {
779
904
  };
780
905
  return getText(c);
781
906
  }).filter(t => t != '').join(config.newlineDelimiter);
782
- return (0, astUtils_js_1.createAST)('pptx', metadata, content, attachments, config, toTextSync);
907
+ const auxiliaryContent = slideMasters.length > 0 ? {
908
+ slideMasters
909
+ } : undefined;
910
+ return (0, astUtils_js_1.createAST)('pptx', metadata, content, attachments, config, auxiliaryContent, toTextSync);
783
911
  };
784
912
  exports.parsePowerPoint = parsePowerPoint;
@@ -359,6 +359,7 @@ exports.SimpleRtfParser = SimpleRtfParser;
359
359
  * @returns The parsed AST.
360
360
  */
361
361
  const parseRtf = async (buffer, config) => {
362
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
362
363
  const parser = new SimpleRtfParser(buffer);
363
364
  const doc = parser.parse();
364
365
  // Extract font and color tables
@@ -1106,11 +1107,14 @@ const parseRtf = async (buffer, config) => {
1106
1107
  }
1107
1108
  // Handle footnote: switch target to notes
1108
1109
  const previousTarget = currentTarget;
1110
+ let savedParagraphTextChunks;
1111
+ let savedParagraphChildren;
1112
+ let savedParagraphRawChunks;
1109
1113
  if (isFootnote) {
1110
1114
  if (config.ignoreNotes) {
1111
1115
  return; // Skip footnote content entirely
1112
1116
  }
1113
- flushParagraph();
1117
+ flushRun();
1114
1118
  currentFootnoteId++;
1115
1119
  // Determine note type based on \fet value
1116
1120
  let noteType = 'footnote';
@@ -1134,8 +1138,25 @@ const parseRtf = async (buffer, config) => {
1134
1138
  noteType: noteType
1135
1139
  }
1136
1140
  };
1137
- notes.push(noteNode);
1141
+ if (currentParagraphChildren.length > 0) {
1142
+ const precedingNode = currentParagraphChildren[currentParagraphChildren.length - 1];
1143
+ if (!precedingNode.notes)
1144
+ precedingNode.notes = [];
1145
+ precedingNode.notes.push(noteNode);
1146
+ }
1147
+ else {
1148
+ const emptyTextNode = { type: 'text', text: '' };
1149
+ emptyTextNode.notes = [noteNode];
1150
+ currentParagraphChildren.push(emptyTextNode);
1151
+ }
1138
1152
  currentTarget = noteNode.children;
1153
+ // Save current paragraph state so we don't mix footnote paragraphs with main text
1154
+ savedParagraphTextChunks = [...currentParagraphTextChunks];
1155
+ savedParagraphChildren = [...currentParagraphChildren];
1156
+ savedParagraphRawChunks = [...currentParagraphRawChunks];
1157
+ currentParagraphTextChunks = [];
1158
+ currentParagraphChildren = [];
1159
+ currentParagraphRawChunks = [];
1139
1160
  }
1140
1161
  // Create a new formatting context for the group
1141
1162
  const groupFormatting = { ...formatting };
@@ -1194,6 +1215,10 @@ const parseRtf = async (buffer, config) => {
1194
1215
  if (isFootnote) {
1195
1216
  flushParagraph();
1196
1217
  currentTarget = previousTarget;
1218
+ // Restore the saved paragraph state
1219
+ currentParagraphTextChunks = savedParagraphTextChunks;
1220
+ currentParagraphChildren = savedParagraphChildren;
1221
+ currentParagraphRawChunks = savedParagraphRawChunks;
1197
1222
  }
1198
1223
  // Clear link URL after processing the field group
1199
1224
  if (isHyperlinkField) {
@@ -1638,21 +1663,10 @@ const parseRtf = async (buffer, config) => {
1638
1663
  flushTable();
1639
1664
  }
1640
1665
  flushParagraph();
1641
- // Notes handling:
1642
- // - If putNotesAtLast is false, notes should be added inline during traversal
1643
- // (currently they go to 'notes' array, then we append them here - this is wrong)
1644
- // - If putNotesAtLast is true, notes are appended at the very end (see below)
1645
- //
1646
- // For now, when putNotesAtLast is false, we append notes immediately after content
1647
- // This isn't truly "inline" but it's better than at the end
1648
- // TODO: Implement true inline placement during traversal
1649
- if (!config.putNotesAtLast && notes.length > 0) {
1650
- content.push(...notes);
1651
- notes.length = 0; // Clear so they don't get appended again
1652
- }
1653
1666
  // Perform OCR if enabled
1654
1667
  if (config.ocr && config.extractAttachments) {
1655
1668
  for (const attachment of attachments) {
1669
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
1656
1670
  if (attachment.mimeType.startsWith('image/')) {
1657
1671
  try {
1658
1672
  // Convert base64 data back to Buffer for Tesseract.js
@@ -1710,20 +1724,12 @@ const parseRtf = async (buffer, config) => {
1710
1724
  populateNoteText(content);
1711
1725
  populateNoteText(notes);
1712
1726
  const toTextSync = () => {
1713
- let text = content.map(c => c.text).join(config.newlineDelimiter);
1714
- if (config.putNotesAtLast && notes.length > 0) {
1715
- text += config.newlineDelimiter + notes.map(c => c.text).join(config.newlineDelimiter);
1716
- }
1717
- return text;
1727
+ return content.map(c => c.text).join(config.newlineDelimiter);
1718
1728
  };
1719
1729
  const result = (0, astUtils_js_1.createAST)('rtf', {
1720
1730
  // RTF Limitation: No style map available (RTF uses inline styles)
1721
1731
  }, content, attachments, // PNG and JPEG images extracted from \\pict groups
1722
- config, toTextSync);
1723
- // If putNotesAtLast is true, append notes to the end of the content array
1724
- if (config.putNotesAtLast && notes.length > 0) {
1725
- content.push(...notes);
1726
- }
1732
+ config, undefined, toTextSync);
1727
1733
  return result;
1728
1734
  };
1729
1735
  exports.parseRtf = parseRtf;