officeparser 6.1.0 → 7.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (70) hide show
  1. package/README.md +284 -86
  2. package/dist/OfficeConverter.d.ts +46 -0
  3. package/dist/OfficeConverter.js +72 -0
  4. package/dist/OfficeGenerator.d.ts +19 -0
  5. package/dist/OfficeGenerator.js +48 -0
  6. package/dist/OfficeParser.d.ts +6 -0
  7. package/dist/OfficeParser.js +55 -28
  8. package/dist/cli.d.ts +3 -1
  9. package/dist/cli.js +107 -22
  10. package/dist/defaults.d.ts +41 -0
  11. package/dist/defaults.js +172 -0
  12. package/dist/generators/BaseGenerator.d.ts +58 -0
  13. package/dist/generators/BaseGenerator.js +107 -0
  14. package/dist/generators/ChunkingGenerator.d.ts +81 -0
  15. package/dist/generators/ChunkingGenerator.js +683 -0
  16. package/dist/generators/CsvGenerator.d.ts +30 -0
  17. package/dist/generators/CsvGenerator.js +233 -0
  18. package/dist/generators/HtmlGenerator.d.ts +37 -0
  19. package/dist/generators/HtmlGenerator.js +1013 -0
  20. package/dist/generators/MarkdownGenerator.d.ts +59 -0
  21. package/dist/generators/MarkdownGenerator.js +481 -0
  22. package/dist/generators/PdfGenerator.d.ts +22 -0
  23. package/dist/generators/PdfGenerator.js +118 -0
  24. package/dist/generators/RtfGenerator.d.ts +15 -0
  25. package/dist/generators/RtfGenerator.js +208 -0
  26. package/dist/generators/TextGenerator.d.ts +13 -0
  27. package/dist/generators/TextGenerator.js +108 -0
  28. package/dist/index.d.ts +11 -3
  29. package/dist/index.js +17 -2
  30. package/dist/index.mjs +2 -2
  31. package/dist/officeparser.browser.d.ts +878 -5
  32. package/dist/officeparser.browser.iife.js +703 -49
  33. package/dist/officeparser.browser.mjs +703 -49
  34. package/dist/parsers/CsvParser.d.ts +9 -0
  35. package/dist/parsers/CsvParser.js +110 -0
  36. package/dist/parsers/ExcelParser.d.ts +2 -2
  37. package/dist/parsers/ExcelParser.js +145 -114
  38. package/dist/parsers/HtmlParser.d.ts +2 -0
  39. package/dist/parsers/HtmlParser.js +539 -0
  40. package/dist/parsers/MarkdownParser.d.ts +2 -0
  41. package/dist/parsers/MarkdownParser.js +360 -0
  42. package/dist/parsers/OpenOfficeParser.d.ts +2 -2
  43. package/dist/parsers/OpenOfficeParser.js +237 -128
  44. package/dist/parsers/PdfParser.d.ts +2 -2
  45. package/dist/parsers/PdfParser.js +52 -49
  46. package/dist/parsers/PowerPointParser.d.ts +2 -2
  47. package/dist/parsers/PowerPointParser.js +132 -123
  48. package/dist/parsers/RtfParser.d.ts +22 -2
  49. package/dist/parsers/RtfParser.js +1398 -1282
  50. package/dist/parsers/WordParser.d.ts +3 -2
  51. package/dist/parsers/WordParser.js +333 -115
  52. package/dist/sbom.cdx.json +103 -103
  53. package/dist/types.d.ts +833 -5
  54. package/dist/types.js +71 -0
  55. package/dist/utils/astUtils.d.ts +16 -0
  56. package/dist/utils/astUtils.js +32 -0
  57. package/dist/utils/configUtils.d.ts +26 -0
  58. package/dist/utils/configUtils.js +140 -0
  59. package/dist/utils/envUtils.js +56 -2
  60. package/dist/utils/errorUtils.d.ts +17 -29
  61. package/dist/utils/errorUtils.js +109 -52
  62. package/dist/utils/moduleLoader.js +15 -9
  63. package/dist/utils/ocrUtils.js +2 -1
  64. package/dist/utils/sheetUtils.d.ts +7 -0
  65. package/dist/utils/sheetUtils.js +35 -0
  66. package/dist/utils/styleMapper.d.ts +36 -0
  67. package/dist/utils/styleMapper.js +224 -0
  68. package/dist/utils/xmlUtils.d.ts +0 -8
  69. package/dist/utils/xmlUtils.js +2 -1
  70. package/package.json +28 -9
@@ -23,6 +23,8 @@
23
23
  */
24
24
  Object.defineProperty(exports, "__esModule", { value: true });
25
25
  exports.parseOpenOffice = void 0;
26
+ const types_js_1 = require("../types.js");
27
+ const astUtils_js_1 = require("../utils/astUtils.js");
26
28
  const chartUtils_js_1 = require("../utils/chartUtils.js");
27
29
  const errorUtils_js_1 = require("../utils/errorUtils.js");
28
30
  const imageUtils_js_1 = require("../utils/imageUtils.js");
@@ -63,6 +65,7 @@ const parseOpenOffice = async (buffer, config) => {
63
65
  }
64
66
  const mainContentFile = files.find(f => f.path === 'content.xml') || files.find(f => f.path.match(contentFileRegex));
65
67
  const stylesFile = files.find(f => f.path === 'styles.xml');
68
+ const stylesDom = stylesFile ? (0, xmlUtils_js_1.parseXmlString)(stylesFile.content.toString()) : undefined;
66
69
  const content = [];
67
70
  const notes = [];
68
71
  // Style Map: styleName -> TextFormatting
@@ -70,9 +73,13 @@ const parseOpenOffice = async (buffer, config) => {
70
73
  const styleMap = {};
71
74
  const paragraphStyleMap = {};
72
75
  const listCounters = {}; // Track item index per listId/level
76
+ let currentListId = null;
77
+ let lastListType = null;
78
+ let lastListStyle = null;
79
+ let listIdCounter = 0;
80
+ let lastWasList = false;
73
81
  // Helper to parse styles
74
- const parseStyles = (xmlString) => {
75
- const xml = (0, xmlUtils_js_1.parseXmlString)(xmlString);
82
+ const parseStyles = (xml) => {
76
83
  const styles = (0, xmlUtils_js_1.getElementsByTagName)(xml, "style:style");
77
84
  for (const style of styles) {
78
85
  const name = style.getAttribute("style:name");
@@ -151,8 +158,8 @@ const parseOpenOffice = async (buffer, config) => {
151
158
  }
152
159
  }
153
160
  };
154
- if (stylesFile) {
155
- parseStyles(stylesFile.content.toString());
161
+ if (stylesDom) {
162
+ parseStyles(stylesDom);
156
163
  }
157
164
  /**
158
165
  * Helper to parse a paragraph node (text:p or text:h) and extract its content.
@@ -172,9 +179,10 @@ const parseOpenOffice = async (buffer, config) => {
172
179
  */
173
180
  const parseInlineContent = (node, styleMap, config, notes, paragraphStyleMap, parentFormatting = {}, linkMetadata, sourceXml = '') => {
174
181
  const children = [];
182
+ const anchorIds = [];
175
183
  let fullText = '';
176
184
  if (!node.childNodes)
177
- return { text: '', children: [] };
185
+ return { text: '', children: [], anchorIds: [] };
178
186
  for (let i = 0; i < node.childNodes.length; i++) {
179
187
  const child = node.childNodes[i];
180
188
  if (child.nodeType === 3) { // Text node
@@ -192,7 +200,12 @@ const parseOpenOffice = async (buffer, config) => {
192
200
  else if ((0, xmlUtils_js_1.isElement)(child)) {
193
201
  const element = child;
194
202
  const tagName = element.tagName;
195
- if (tagName === 'text:s') {
203
+ if (tagName === 'text:bookmark' || tagName === 'text:bookmark-start') {
204
+ const name = element.getAttribute('text:name');
205
+ if (name)
206
+ anchorIds.push(name);
207
+ }
208
+ else if (tagName === 'text:s') {
196
209
  // Space
197
210
  const count = parseInt(element.getAttribute('text:c') || '1');
198
211
  const spaces = ' '.repeat(count);
@@ -221,7 +234,7 @@ const parseOpenOffice = async (buffer, config) => {
221
234
  type: 'text',
222
235
  text: '\n',
223
236
  formatting: parentFormatting,
224
- metadata: linkMetadata ? { ...linkMetadata } : undefined
237
+ metadata: { ...(linkMetadata || {}), isLineBreak: true }
225
238
  });
226
239
  }
227
240
  else if (tagName === 'text:span') {
@@ -231,15 +244,34 @@ const parseOpenOffice = async (buffer, config) => {
231
244
  const spanContent = parseInlineContent(element, styleMap, config, notes, paragraphStyleMap, formatting, linkMetadata, sourceXml);
232
245
  fullText += spanContent.text;
233
246
  children.push(...spanContent.children);
247
+ anchorIds.push(...spanContent.anchorIds);
234
248
  }
235
249
  else if (tagName === 'text:a') {
236
250
  // Hyperlink
237
- const href = element.getAttribute('xlink:href') || '';
238
- const linkType = href.startsWith('#') ? 'internal' : 'external';
239
- const newLinkMetadata = { link: href, linkType: linkType };
251
+ let href = element.getAttribute('xlink:href') || '';
252
+ const isInternal = href.startsWith('#');
253
+ const linkType = isInternal ? 'internal' : 'external';
254
+ if (isInternal) {
255
+ // ODT internal links can be encoded and might have suffixes like |outline
256
+ try {
257
+ href = decodeURIComponent(href).split('|')[0];
258
+ }
259
+ catch (e) {
260
+ href = href.split('|')[0];
261
+ }
262
+ // Normalize internal link: if it contains #, keep only from # onwards
263
+ if (href.includes('#')) {
264
+ href = '#' + href.split('#').pop();
265
+ }
266
+ }
267
+ let newLinkMetadata;
268
+ if (!isInternal || !config.ignoreInternalLinks) {
269
+ newLinkMetadata = { link: href, linkType: linkType };
270
+ }
240
271
  const linkContent = parseInlineContent(element, styleMap, config, notes, paragraphStyleMap, parentFormatting, newLinkMetadata, sourceXml);
241
272
  fullText += linkContent.text;
242
273
  children.push(...linkContent.children);
274
+ anchorIds.push(...linkContent.anchorIds);
243
275
  }
244
276
  else if (tagName === 'text:note' && !config.ignoreNotes) {
245
277
  // Footnote or endnote
@@ -258,7 +290,10 @@ const parseOpenOffice = async (buffer, config) => {
258
290
  type: 'paragraph',
259
291
  text: npContent.text,
260
292
  children: npContent.children,
261
- metadata: npContent.alignment ? { alignment: npContent.alignment } : undefined
293
+ metadata: {
294
+ ...(npContent.alignment ? { alignment: npContent.alignment } : {}),
295
+ ...(npContent.anchorIds?.length ? { anchorIds: npContent.anchorIds } : {})
296
+ }
262
297
  };
263
298
  noteChildren.push(npNode);
264
299
  }
@@ -318,7 +353,7 @@ const parseOpenOffice = async (buffer, config) => {
318
353
  }
319
354
  }
320
355
  }
321
- return { text: fullText, children };
356
+ return { text: fullText, children, anchorIds };
322
357
  };
323
358
  /**
324
359
  * Helper to parse a paragraph node (text:p or text:h) and extract its content.
@@ -389,7 +424,32 @@ const parseOpenOffice = async (buffer, config) => {
389
424
  }
390
425
  }
391
426
  }
392
- return { text: content.text, children: content.children, alignment, style: paraStyle || undefined };
427
+ return { text: content.text, children: content.children, alignment, style: paraStyle || undefined, anchorIds: content.anchorIds };
428
+ };
429
+ /**
430
+ * Splits paragraph content into multiple segments based on line breaks.
431
+ * Used to handle soft line breaks within list items.
432
+ *
433
+ * @param pContent - The content of a single paragraph
434
+ * @returns Array of content segments
435
+ */
436
+ const splitParagraphByBreaks = (pContent) => {
437
+ const segments = [];
438
+ let currentText = "";
439
+ let currentChildren = [];
440
+ for (const child of pContent.children) {
441
+ if (child.type === "text" && child.metadata?.isLineBreak) {
442
+ segments.push({ text: currentText, children: currentChildren });
443
+ currentText = "";
444
+ currentChildren = [];
445
+ }
446
+ else {
447
+ currentText += child.text || "";
448
+ currentChildren.push(child);
449
+ }
450
+ }
451
+ segments.push({ text: currentText, children: currentChildren });
452
+ return segments;
393
453
  };
394
454
  /**
395
455
  * Helper to parse a table node and extract its structure.
@@ -603,8 +663,9 @@ const parseOpenOffice = async (buffer, config) => {
603
663
  (0, xmlUtils_js_1.getDirectChildren)(officeBody, "office:spreadsheet")[0];
604
664
  if (bodyContent) {
605
665
  const bodyChildren = (0, xmlUtils_js_1.getDirectChildren)(bodyContent, "*");
666
+ const isSpreadsheet = bodyContent.tagName === "office:spreadsheet";
606
667
  for (const child of bodyChildren) {
607
- traverse(child, content, false, xmlString);
668
+ traverse(child, content, false, xmlString, isSpreadsheet);
608
669
  }
609
670
  }
610
671
  }
@@ -616,19 +677,28 @@ const parseOpenOffice = async (buffer, config) => {
616
677
  * @param targetArray - The array to push extracted content nodes to
617
678
  * @param forceHeading - If true, treats all paragraphs as headings (used for slide titles)
618
679
  * @param sourceXml - The source XML string for raw content extraction
680
+ * @param asSheet - If true, treats tables as sheets (for ODS)
619
681
  */
620
- function traverse(node, targetArray, forceHeading = false, sourceXml) {
682
+ function traverse(node, targetArray, forceHeading = false, sourceXml, asSheet = false) {
621
683
  if (node.tagName === "text:p") {
622
684
  const pContent = parseParagraphContent(node, paragraphStyleMap, styleMap, config, sourceXml);
623
685
  const type = (forceHeading || (node.getAttribute("text:style-name") || '').toLowerCase().includes('title')) ? 'heading' : 'paragraph';
686
+ const metadata = {
687
+ ...(pContent.alignment ? { alignment: pContent.alignment } : {}),
688
+ ...(pContent.style ? { style: pContent.style } : {}),
689
+ ...(pContent.anchorIds?.length ? { anchorIds: pContent.anchorIds } : {})
690
+ };
691
+ const nodeId = node.getAttribute("xml:id") || node.getAttribute("text:id");
692
+ if (nodeId) {
693
+ if (!metadata.anchorIds)
694
+ metadata.anchorIds = [];
695
+ metadata.anchorIds.push(nodeId);
696
+ }
624
697
  const pNode = {
625
698
  type,
626
699
  text: pContent.text,
627
700
  children: pContent.children,
628
- metadata: {
629
- ...(pContent.alignment ? { alignment: pContent.alignment } : {}),
630
- ...(pContent.style ? { style: pContent.style } : {})
631
- }
701
+ metadata
632
702
  };
633
703
  if (type === 'heading' && pNode.metadata) {
634
704
  pNode.metadata.level = pNode.metadata.level || 1;
@@ -640,42 +710,65 @@ const parseOpenOffice = async (buffer, config) => {
640
710
  pNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
641
711
  }
642
712
  targetArray.push(pNode);
713
+ lastWasList = false;
643
714
  }
644
715
  else if (node.tagName === "text:h") {
645
716
  const level = parseInt(node.getAttribute("text:outline-level") || "1");
646
717
  const hContent = parseParagraphContent(node, paragraphStyleMap, styleMap, config, sourceXml);
718
+ const metadata = {
719
+ level,
720
+ ...(hContent.alignment ? { alignment: hContent.alignment } : {}),
721
+ ...(hContent.style ? { style: hContent.style } : {}),
722
+ ...(hContent.anchorIds?.length ? { anchorIds: hContent.anchorIds } : {})
723
+ };
724
+ const nodeId = node.getAttribute("xml:id") || node.getAttribute("text:id");
725
+ if (nodeId) {
726
+ if (!metadata.anchorIds)
727
+ metadata.anchorIds = [];
728
+ metadata.anchorIds.push(nodeId);
729
+ }
647
730
  const hNode = {
648
731
  type: 'heading',
649
732
  text: hContent.text,
650
733
  children: hContent.children,
651
- metadata: {
652
- level,
653
- ...(hContent.alignment ? { alignment: hContent.alignment } : {}),
654
- ...(hContent.style ? { style: hContent.style } : {})
655
- }
734
+ metadata
656
735
  };
657
736
  if (config.includeRawContent) {
658
737
  hNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
659
738
  }
660
739
  targetArray.push(hNode);
740
+ lastWasList = false;
661
741
  }
662
742
  else if (node.tagName === "table:table") {
663
743
  // Parse table with proper structure
664
744
  const tableNode = parseTable(node, paragraphStyleMap, styleMap, config, sourceXml);
745
+ if (asSheet) {
746
+ tableNode.type = 'sheet';
747
+ const sheetName = node.getAttribute("table:name");
748
+ if (sheetName) {
749
+ tableNode.metadata = { ...tableNode.metadata, sheetName };
750
+ }
751
+ }
752
+ const tableId = node.getAttribute("xml:id") || node.getAttribute("table:name");
753
+ if (tableId) {
754
+ if (!tableNode.metadata)
755
+ tableNode.metadata = {};
756
+ tableNode.metadata.anchorIds = tableNode.metadata.anchorIds || [];
757
+ tableNode.metadata.anchorIds.push(tableId);
758
+ }
665
759
  if (config.includeRawContent) {
666
760
  tableNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
667
761
  }
668
762
  targetArray.push(tableNode);
763
+ lastWasList = false;
669
764
  }
670
765
  else if (node.tagName === "text:list") {
671
766
  // Parse list structure with proper listId tracking
672
767
  const listItems = (0, xmlUtils_js_1.getDirectChildren)(node, "text:list-item");
673
- // Get list style name to use as listId (or generate one)
674
- const listStyleName = node.getAttribute("text:style-name") || node.getAttribute("xml:id");
675
- const listId = listStyleName || `list-${targetArray.length}`;
676
768
  // Determine list type by checking the list style definition
677
769
  let listType = 'unordered';
678
770
  let isVisible = false;
771
+ const listStyleName = node.getAttribute("text:style-name") || node.getAttribute("xml:id");
679
772
  let styleNameToCheck = listStyleName;
680
773
  // If no style name, check parent list for inherited style
681
774
  if (!styleNameToCheck) {
@@ -689,9 +782,8 @@ const parseOpenOffice = async (buffer, config) => {
689
782
  parentNode = parentNode.parentNode;
690
783
  }
691
784
  }
692
- // Try to find list style in automatic styles to determine type and visibility
785
+ // Try to find list style in automatic styles or styles.xml to determine type and visibility
693
786
  if (styleNameToCheck) {
694
- const automaticStyles = (0, xmlUtils_js_1.getFirstElementByTagName)((0, xmlUtils_js_1.parseXmlString)(mainContentFile?.content.toString() || ''), "office:automatic-styles");
695
787
  if (automaticStyles) {
696
788
  const listStyles = (0, xmlUtils_js_1.getElementsByTagName)(automaticStyles, "text:list-style");
697
789
  for (const listStyle of listStyles) {
@@ -708,32 +800,32 @@ const parseOpenOffice = async (buffer, config) => {
708
800
  listType = 'unordered';
709
801
  isVisible = bulletLevels.some(l => !!l.getAttribute("text:bullet-char"));
710
802
  }
711
- if (imageLevels.length > 0)
803
+ else if (imageLevels.length > 0) {
804
+ listType = 'unordered';
712
805
  isVisible = true;
806
+ }
713
807
  break;
714
808
  }
715
809
  }
716
810
  }
717
- // Also check in styles.xml if still unordered and hidden
718
- if (stylesFile && !isVisible) {
719
- const stylesXml = (0, xmlUtils_js_1.parseXmlString)(stylesFile.content.toString());
720
- const listStyles = (0, xmlUtils_js_1.getElementsByTagName)(stylesXml, "text:list-style");
721
- for (const listStyle of listStyles) {
722
- if (listStyle.getAttribute("style:name") === styleNameToCheck) {
723
- const bulletLevels = (0, xmlUtils_js_1.getElementsByTagName)(listStyle, "text:list-level-style-bullet");
724
- const numberLevels = (0, xmlUtils_js_1.getElementsByTagName)(listStyle, "text:list-level-style-number");
725
- const imageLevels = (0, xmlUtils_js_1.getElementsByTagName)(listStyle, "text:list-level-style-image");
726
- if (numberLevels.length > 0) {
727
- listType = 'ordered';
728
- isVisible = numberLevels.some(l => !!l.getAttribute("style:num-format"));
729
- }
730
- else if (bulletLevels.length > 0) {
731
- listType = 'unordered';
732
- isVisible = bulletLevels.some(l => !!l.getAttribute("text:bullet-char"));
811
+ if (!isVisible && stylesDom) {
812
+ const officeStyles = (0, xmlUtils_js_1.getFirstElementByTagName)(stylesDom, "office:styles");
813
+ if (officeStyles) {
814
+ const listStyles = (0, xmlUtils_js_1.getElementsByTagName)(officeStyles, "text:list-style");
815
+ for (const listStyle of listStyles) {
816
+ if (listStyle.getAttribute("style:name") === styleNameToCheck) {
817
+ const bulletLevels = (0, xmlUtils_js_1.getElementsByTagName)(listStyle, "text:list-level-style-bullet");
818
+ const numberLevels = (0, xmlUtils_js_1.getElementsByTagName)(listStyle, "text:list-level-style-number");
819
+ if (numberLevels.length > 0) {
820
+ listType = 'ordered';
821
+ isVisible = numberLevels.some(l => !!l.getAttribute("style:num-format"));
822
+ }
823
+ else if (bulletLevels.length > 0) {
824
+ listType = 'unordered';
825
+ isVisible = bulletLevels.some(l => !!l.getAttribute("text:bullet-char"));
826
+ }
827
+ break;
733
828
  }
734
- if (imageLevels.length > 0)
735
- isVisible = true;
736
- break;
737
829
  }
738
830
  }
739
831
  }
@@ -741,6 +833,7 @@ const parseOpenOffice = async (buffer, config) => {
741
833
  // If the list is not visible, it's likely a layout list used by Impress.
742
834
  // We should traverse its items and treat their content as regular nodes.
743
835
  if (!isVisible) {
836
+ lastWasList = false;
744
837
  for (let i = 0; i < listItems.length; i++) {
745
838
  const item = listItems[i];
746
839
  if (item.childNodes) {
@@ -754,6 +847,24 @@ const parseOpenOffice = async (buffer, config) => {
754
847
  }
755
848
  return;
756
849
  }
850
+ // List Continuity Logic:
851
+ // If this list follows another list of the same type and style, or we are in ODP and it's sequential,
852
+ // we should reuse the previous listId to maintain numbering.
853
+ const isODP = fileType === 'odp';
854
+ const sameStyle = styleNameToCheck && styleNameToCheck === lastListStyle;
855
+ const sameType = listType === lastListType;
856
+ let listId;
857
+ if (lastWasList && (sameStyle || (isODP && sameType))) {
858
+ listId = currentListId;
859
+ }
860
+ else {
861
+ // New list
862
+ listId = styleNameToCheck || `list-${++listIdCounter}`;
863
+ currentListId = listId;
864
+ lastListType = listType;
865
+ lastListStyle = styleNameToCheck;
866
+ }
867
+ lastWasList = true;
757
868
  // Calculate indentation level by counting parent text:list elements
758
869
  let indentation = 0;
759
870
  let parent = node.parentNode;
@@ -774,59 +885,57 @@ const parseOpenOffice = async (buffer, config) => {
774
885
  // Process each list item
775
886
  for (let i = 0; i < listItems.length; i++) {
776
887
  const item = listItems[i];
777
- // Increment item index for this list/level
778
- listCounters[listId][indentKey]++;
779
- const itemIndex = listCounters[listId][indentKey];
780
- // Reset deeper levels when we encounter an item at this level
781
- for (let k = indentation + 1; k < 10; k++) {
782
- if (listCounters[listId][k.toString()] !== undefined) {
783
- listCounters[listId][k.toString()] = -1;
784
- }
785
- }
888
+ let hasIndexedThisItem = false;
786
889
  // Iterate over direct children of list item (paragraphs, headings, nested lists)
787
890
  if (item.childNodes) {
788
891
  for (let j = 0; j < item.childNodes.length; j++) {
789
892
  const child = item.childNodes[j];
790
893
  if ((0, xmlUtils_js_1.isElement)(child)) { // Element
791
894
  const element = child;
792
- if (element.tagName === "text:p") {
793
- const pContent = parseParagraphContent(element, paragraphStyleMap, styleMap, config, sourceXml);
794
- const listNode = {
795
- type: 'list',
796
- text: pContent.text,
797
- children: pContent.children,
798
- metadata: {
799
- listType,
800
- indentation,
801
- itemIndex,
802
- listId,
803
- alignment: pContent.alignment || 'left',
804
- style: pContent.style
895
+ if (element.tagName === "text:p" || element.tagName === "text:h") {
896
+ if (!hasIndexedThisItem) {
897
+ listCounters[listId][indentKey]++;
898
+ hasIndexedThisItem = true;
899
+ for (let k = indentation + 1; k < 10; k++) {
900
+ if (listCounters[listId][k.toString()] !== undefined) {
901
+ listCounters[listId][k.toString()] = -1;
902
+ }
805
903
  }
806
- };
807
- if (config.includeRawContent)
808
- listNode.rawContent = (0, xmlUtils_js_1.getRawContent)(element, sourceXml, config);
809
- targetArray.push(listNode);
810
- }
811
- else if (element.tagName === "text:h") {
812
- const level = parseInt(element.getAttribute("text:outline-level") || "1");
813
- const hContent = parseParagraphContent(element, paragraphStyleMap, styleMap, config, sourceXml);
814
- const listNode = {
815
- type: 'list',
816
- text: hContent.text,
817
- children: hContent.children,
818
- metadata: {
819
- listType,
820
- indentation,
821
- itemIndex,
822
- listId,
823
- ...(hContent.alignment ? { alignment: hContent.alignment } : {}),
824
- style: hContent.style
904
+ }
905
+ const itemIndex = listCounters[listId][indentKey];
906
+ const pContent = parseParagraphContent(element, paragraphStyleMap, styleMap, config, sourceXml);
907
+ const segments = splitParagraphByBreaks(pContent);
908
+ for (let k = 0; k < segments.length; k++) {
909
+ const segment = segments[k];
910
+ if (!segment.text.trim() && segment.children.length === 0)
911
+ continue;
912
+ const isFirst = k === 0;
913
+ const nodeType = isFirst ? 'list' : 'paragraph';
914
+ const node = {
915
+ type: nodeType,
916
+ text: segment.text,
917
+ children: segment.children,
918
+ metadata: isFirst ? {
919
+ listType,
920
+ indentation,
921
+ itemIndex,
922
+ listId,
923
+ alignment: pContent.alignment || 'left',
924
+ style: pContent.style
925
+ } : {
926
+ alignment: pContent.alignment || 'left',
927
+ style: pContent.style
928
+ }
929
+ };
930
+ // Special case for headings in lists
931
+ if (isFirst && element.tagName === "text:h") {
932
+ const level = parseInt(element.getAttribute("text:outline-level") || "1");
933
+ node.metadata.level = level;
825
934
  }
826
- };
827
- if (config.includeRawContent)
828
- listNode.rawContent = (0, xmlUtils_js_1.getRawContent)(element, sourceXml, config);
829
- targetArray.push(listNode);
935
+ if (config.includeRawContent)
936
+ node.rawContent = (0, xmlUtils_js_1.getRawContent)(element, sourceXml, config);
937
+ targetArray.push(node);
938
+ }
830
939
  }
831
940
  else if (element.tagName === "text:list") {
832
941
  // Recursive call for nested list
@@ -871,14 +980,19 @@ const parseOpenOffice = async (buffer, config) => {
871
980
  const parts = imageHref.split('/');
872
981
  imageHref = parts[parts.length - 1];
873
982
  }
983
+ const metadata = {
984
+ attachmentName: imageHref,
985
+ ...(altText ? { altText } : {})
986
+ };
987
+ const frameId = node.getAttribute("xml:id") || node.getAttribute("draw:name");
988
+ if (frameId) {
989
+ metadata.anchorIds = [frameId];
990
+ }
874
991
  const imageNode = {
875
992
  type: 'image',
876
993
  text: '',
877
994
  children: [],
878
- metadata: {
879
- attachmentName: imageHref,
880
- ...(altText ? { altText } : {})
881
- }
995
+ metadata
882
996
  };
883
997
  if (config.includeRawContent) {
884
998
  imageNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
@@ -1211,10 +1325,10 @@ const parseOpenOffice = async (buffer, config) => {
1211
1325
  if (config.ocr) {
1212
1326
  if (attachment.mimeType.startsWith('image/')) {
1213
1327
  try {
1214
- attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(media.content, { language: config.ocrLanguage, ...config.ocrConfig })).trim();
1328
+ attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(media.content, { ...config.ocrConfig })).trim();
1215
1329
  }
1216
1330
  catch (e) {
1217
- (0, errorUtils_js_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
1331
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.OCR_FAILED, config, attachment.name, e);
1218
1332
  }
1219
1333
  }
1220
1334
  }
@@ -1351,7 +1465,7 @@ const parseOpenOffice = async (buffer, config) => {
1351
1465
  node.text = attachment.ocrText;
1352
1466
  }
1353
1467
  if (attachment.chartData && node.type === 'chart') {
1354
- node.text = attachment.chartData.rawTexts.join(config.newlineDelimiter ?? '\n');
1468
+ node.text = attachment.chartData.rawTexts.join(config.newlineDelimiter);
1355
1469
  }
1356
1470
  }
1357
1471
  }
@@ -1385,32 +1499,27 @@ const parseOpenOffice = async (buffer, config) => {
1385
1499
  if (config.putNotesAtLast && notes.length > 0) {
1386
1500
  content.push(...notes);
1387
1501
  }
1388
- return {
1389
- type: fileType,
1390
- metadata: {
1391
- ...metadata,
1392
- styleMap: combinedStyleMap
1393
- },
1394
- content: content,
1395
- attachments: attachments,
1396
- toText: () => content.map(c => {
1397
- const getText = (node) => {
1398
- let t = '';
1399
- if (node.children && node.children.length > 0) {
1400
- // Check if children have their own children (container vs leaf)
1401
- // If children are leaf nodes (text/image), join with empty string
1402
- // If children are container nodes (paragraphs/rows), join with newline
1403
- const hasGrandChildren = node.children.some(child => child.children && child.children.length > 0);
1404
- const separator = hasGrandChildren ? (config.newlineDelimiter ?? '\n') : '';
1405
- t += node.children.map(getText).filter(t => t != '').join(separator);
1406
- }
1407
- else {
1408
- t += node.text || '';
1409
- }
1410
- return t;
1411
- };
1412
- return getText(c);
1413
- }).filter(t => t != '').join(config.newlineDelimiter ?? '\n')
1414
- };
1502
+ const toTextSync = () => content.map(c => {
1503
+ const getText = (node) => {
1504
+ let t = '';
1505
+ if (node.children && node.children.length > 0) {
1506
+ // Check if children have their own children (container vs leaf)
1507
+ // If children are leaf nodes (text/image), join with empty string
1508
+ // If children are container nodes (paragraphs/rows), join with newline
1509
+ const hasGrandChildren = node.children.some(child => child.children && child.children.length > 0);
1510
+ const separator = hasGrandChildren ? config.newlineDelimiter : '';
1511
+ t += node.children.map(getText).filter(t => t != '').join(separator);
1512
+ }
1513
+ else {
1514
+ t += node.text || '';
1515
+ }
1516
+ return t;
1517
+ };
1518
+ return getText(c);
1519
+ }).filter(t => t != '').join(config.newlineDelimiter);
1520
+ return (0, astUtils_js_1.createAST)(fileType, {
1521
+ ...metadata,
1522
+ styleMap: combinedStyleMap
1523
+ }, content, attachments, config, toTextSync);
1415
1524
  };
1416
1525
  exports.parseOpenOffice = parseOpenOffice;
@@ -56,7 +56,7 @@
56
56
  * @see https://mozilla.github.io/pdf.js/ PDF.js documentation
57
57
  * @see https://www.adobe.com/devnet/pdf/pdf_reference.html PDF Reference
58
58
  */
59
- import { OfficeParserAST, OfficeParserConfig } from '../types.js';
59
+ import { FullOfficeParserConfig, OfficeParserAST } from '../types.js';
60
60
  /**
61
61
  * Parses a PDF file and extracts content.
62
62
  *
@@ -64,4 +64,4 @@ import { OfficeParserAST, OfficeParserConfig } from '../types.js';
64
64
  * @param config - Parser configuration
65
65
  * @returns Promise resolving to the parsed AST
66
66
  */
67
- export declare const parsePdf: (buffer: Buffer, config: OfficeParserConfig) => Promise<OfficeParserAST>;
67
+ export declare const parsePdf: (buffer: Buffer, config: FullOfficeParserConfig) => Promise<OfficeParserAST>;