officeparser 6.0.7 → 6.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/README.md +136 -52
  2. package/dist/OfficeParser.d.ts +10 -1
  3. package/dist/OfficeParser.js +44 -56
  4. package/dist/cli.d.ts +20 -0
  5. package/dist/cli.js +117 -0
  6. package/dist/index.d.ts +4 -4
  7. package/dist/index.js +7 -59
  8. package/dist/index.mjs +18 -0
  9. package/dist/officeparser.browser.d.ts +133 -3
  10. package/dist/officeparser.browser.iife.js +115 -0
  11. package/dist/officeparser.browser.mjs +114 -0
  12. package/dist/parsers/ExcelParser.d.ts +1 -1
  13. package/dist/parsers/ExcelParser.js +76 -68
  14. package/dist/parsers/OpenOfficeParser.d.ts +1 -1
  15. package/dist/parsers/OpenOfficeParser.js +224 -159
  16. package/dist/parsers/PdfParser.d.ts +1 -1
  17. package/dist/parsers/PdfParser.js +98 -94
  18. package/dist/parsers/PowerPointParser.d.ts +1 -1
  19. package/dist/parsers/PowerPointParser.js +188 -179
  20. package/dist/parsers/RtfParser.d.ts +21 -1
  21. package/dist/parsers/RtfParser.js +117 -48
  22. package/dist/parsers/WordParser.d.ts +2 -1
  23. package/dist/parsers/WordParser.js +214 -123
  24. package/dist/sbom.cdx.json +1807 -0
  25. package/dist/types.d.ts +123 -3
  26. package/dist/utils/chartUtils.js +2 -0
  27. package/dist/utils/dateUtils.d.ts +17 -0
  28. package/dist/utils/dateUtils.js +69 -0
  29. package/dist/utils/envUtils.d.ts +24 -0
  30. package/dist/utils/envUtils.js +69 -0
  31. package/dist/utils/moduleLoader.d.ts +2 -1
  32. package/dist/utils/moduleLoader.js +9 -39
  33. package/dist/utils/ocrUtils.d.ts +16 -12
  34. package/dist/utils/ocrUtils.js +186 -25
  35. package/dist/utils/xmlUtils.d.ts +80 -9
  36. package/dist/utils/xmlUtils.js +236 -18
  37. package/dist/utils/zipUtils.js +6 -47
  38. package/package.json +31 -16
  39. package/dist/officeParserBundle@6.0.7.js +0 -154
  40. package/dist/officeparser.browser.js +0 -154
@@ -23,12 +23,12 @@
23
23
  */
24
24
  Object.defineProperty(exports, "__esModule", { value: true });
25
25
  exports.parseOpenOffice = void 0;
26
- const chartUtils_1 = require("../utils/chartUtils");
27
- const errorUtils_1 = require("../utils/errorUtils");
28
- const imageUtils_1 = require("../utils/imageUtils");
29
- const ocrUtils_1 = require("../utils/ocrUtils");
30
- const xmlUtils_1 = require("../utils/xmlUtils");
31
- const zipUtils_1 = require("../utils/zipUtils");
26
+ const chartUtils_js_1 = require("../utils/chartUtils.js");
27
+ const errorUtils_js_1 = require("../utils/errorUtils.js");
28
+ const imageUtils_js_1 = require("../utils/imageUtils.js");
29
+ const ocrUtils_js_1 = require("../utils/ocrUtils.js");
30
+ const xmlUtils_js_1 = require("../utils/xmlUtils.js");
31
+ const zipUtils_js_1 = require("../utils/zipUtils.js");
32
32
  /**
33
33
  * Parses an OpenOffice document (.odt, .odp, .ods) and extracts content.
34
34
  *
@@ -43,7 +43,7 @@ const parseOpenOffice = async (buffer, config) => {
43
43
  const metaFileRegex = /meta\.xml/;
44
44
  const stylesFileRegex = /styles\.xml/;
45
45
  const mimetypeFileRegex = /mimetype/;
46
- const files = await (0, zipUtils_1.extractFiles)(buffer, x => !!x.match(contentFileRegex) ||
46
+ const files = await (0, zipUtils_js_1.extractFiles)(buffer, x => !!x.match(contentFileRegex) ||
47
47
  !!x.match(objectContentFileRegex) ||
48
48
  !!x.match(metaFileRegex) ||
49
49
  !!x.match(stylesFileRegex) ||
@@ -70,17 +70,22 @@ const parseOpenOffice = async (buffer, config) => {
70
70
  const styleMap = {};
71
71
  const paragraphStyleMap = {};
72
72
  const listCounters = {}; // Track item index per listId/level
73
+ let currentListId = null;
74
+ let lastListType = null;
75
+ let lastListStyle = null;
76
+ let listIdCounter = 0;
77
+ let lastWasList = false;
73
78
  // Helper to parse styles
74
79
  const parseStyles = (xmlString) => {
75
- const xml = (0, xmlUtils_1.parseXmlString)(xmlString);
76
- const styles = (0, xmlUtils_1.getElementsByTagName)(xml, "style:style");
80
+ const xml = (0, xmlUtils_js_1.parseXmlString)(xmlString);
81
+ const styles = (0, xmlUtils_js_1.getElementsByTagName)(xml, "style:style");
77
82
  for (const style of styles) {
78
83
  const name = style.getAttribute("style:name");
79
84
  if (!name)
80
85
  continue;
81
86
  const styleInfo = {};
82
87
  // Parse paragraph properties for alignment and drop caps
83
- const paraProps = (0, xmlUtils_1.getElementsByTagName)(style, "style:paragraph-properties")[0];
88
+ const paraProps = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "style:paragraph-properties");
84
89
  if (paraProps) {
85
90
  const textAlign = paraProps.getAttribute("fo:text-align");
86
91
  if (textAlign) {
@@ -97,7 +102,7 @@ const parseOpenOffice = async (buffer, config) => {
97
102
  }
98
103
  }
99
104
  // Detect Drop Caps
100
- const dropCap = (0, xmlUtils_1.getElementsByTagName)(paraProps, "style:drop-cap")[0];
105
+ const dropCap = (0, xmlUtils_js_1.getFirstElementByTagName)(paraProps, "style:drop-cap");
101
106
  if (dropCap) {
102
107
  styleInfo.dropCap = true;
103
108
  }
@@ -106,9 +111,9 @@ const parseOpenOffice = async (buffer, config) => {
106
111
  paragraphStyleMap[name] = styleInfo;
107
112
  }
108
113
  // Parse text properties
109
- const textProps = (0, xmlUtils_1.getElementsByTagName)(style, "style:text-properties")[0];
114
+ const textProps = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "style:text-properties");
110
115
  // Parse table cell properties (for ODS background)
111
- const cellProps = (0, xmlUtils_1.getElementsByTagName)(style, "style:table-cell-properties")[0];
116
+ const cellProps = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "style:table-cell-properties");
112
117
  const formatting = {};
113
118
  if (cellProps) {
114
119
  const bgColor = cellProps.getAttribute("fo:background-color");
@@ -170,7 +175,7 @@ const parseOpenOffice = async (buffer, config) => {
170
175
  * @param linkMetadata - Metadata inherited from parent link
171
176
  * @returns Object containing text and children
172
177
  */
173
- const parseInlineContent = (node, styleMap, config, notes, paragraphStyleMap, parentFormatting = {}, linkMetadata) => {
178
+ const parseInlineContent = (node, styleMap, config, notes, paragraphStyleMap, parentFormatting = {}, linkMetadata, sourceXml = '') => {
174
179
  const children = [];
175
180
  let fullText = '';
176
181
  if (!node.childNodes)
@@ -189,7 +194,7 @@ const parseOpenOffice = async (buffer, config) => {
189
194
  });
190
195
  }
191
196
  }
192
- else if (child.nodeType === 1) {
197
+ else if ((0, xmlUtils_js_1.isElement)(child)) {
193
198
  const element = child;
194
199
  const tagName = element.tagName;
195
200
  if (tagName === 'text:s') {
@@ -221,14 +226,14 @@ const parseOpenOffice = async (buffer, config) => {
221
226
  type: 'text',
222
227
  text: '\n',
223
228
  formatting: parentFormatting,
224
- metadata: linkMetadata ? { ...linkMetadata } : undefined
229
+ metadata: { ...(linkMetadata || {}), isLineBreak: true }
225
230
  });
226
231
  }
227
232
  else if (tagName === 'text:span') {
228
233
  // Formatted text span
229
234
  const styleName = element.getAttribute("text:style-name");
230
235
  const formatting = styleName ? { ...parentFormatting, ...styleMap[styleName] } : parentFormatting;
231
- const spanContent = parseInlineContent(element, styleMap, config, notes, paragraphStyleMap, formatting, linkMetadata);
236
+ const spanContent = parseInlineContent(element, styleMap, config, notes, paragraphStyleMap, formatting, linkMetadata, sourceXml);
232
237
  fullText += spanContent.text;
233
238
  children.push(...spanContent.children);
234
239
  }
@@ -237,7 +242,7 @@ const parseOpenOffice = async (buffer, config) => {
237
242
  const href = element.getAttribute('xlink:href') || '';
238
243
  const linkType = href.startsWith('#') ? 'internal' : 'external';
239
244
  const newLinkMetadata = { link: href, linkType: linkType };
240
- const linkContent = parseInlineContent(element, styleMap, config, notes, paragraphStyleMap, parentFormatting, newLinkMetadata);
245
+ const linkContent = parseInlineContent(element, styleMap, config, notes, paragraphStyleMap, parentFormatting, newLinkMetadata, sourceXml);
241
246
  fullText += linkContent.text;
242
247
  children.push(...linkContent.children);
243
248
  }
@@ -245,14 +250,14 @@ const parseOpenOffice = async (buffer, config) => {
245
250
  // Footnote or endnote
246
251
  const noteClass = (element.getAttribute('text:note-class') || 'footnote');
247
252
  const noteId = element.getAttribute('text:id') || element.getAttribute('xml:id') || undefined;
248
- const noteBody = (0, xmlUtils_1.getElementsByTagName)(element, "text:note-body")[0];
253
+ const noteBody = (0, xmlUtils_js_1.getFirstElementByTagName)(element, "text:note-body");
249
254
  if (noteBody) {
250
255
  // Extract note content recursively
251
- const notePs = (0, xmlUtils_1.getElementsByTagName)(noteBody, "text:p");
256
+ const notePs = (0, xmlUtils_js_1.getElementsByTagName)(noteBody, "text:p");
252
257
  const noteChildren = [];
253
258
  let noteText = '';
254
259
  for (const np of notePs) {
255
- const npContent = parseParagraphContent(np, paragraphStyleMap, styleMap, config);
260
+ const npContent = parseParagraphContent(np, paragraphStyleMap, styleMap, config, sourceXml);
256
261
  noteText += (noteText ? ' ' : '') + npContent.text;
257
262
  const npNode = {
258
263
  type: 'paragraph',
@@ -284,8 +289,8 @@ const parseOpenOffice = async (buffer, config) => {
284
289
  const frame = element;
285
290
  // Extract alt text
286
291
  let altText = '';
287
- const svgTitle = (0, xmlUtils_1.getElementsByTagName)(frame, "svg:title")[0];
288
- const svgDesc = (0, xmlUtils_1.getElementsByTagName)(frame, "svg:desc")[0];
292
+ const svgTitle = (0, xmlUtils_js_1.getFirstElementByTagName)(frame, "svg:title");
293
+ const svgDesc = (0, xmlUtils_js_1.getFirstElementByTagName)(frame, "svg:desc");
289
294
  if (svgTitle && svgTitle.textContent) {
290
295
  altText = svgTitle.textContent;
291
296
  }
@@ -294,7 +299,7 @@ const parseOpenOffice = async (buffer, config) => {
294
299
  }
295
300
  // Extract image href
296
301
  let imageHref = '';
297
- const drawImages = (0, xmlUtils_1.getElementsByTagName)(frame, "draw:image");
302
+ const drawImages = (0, xmlUtils_js_1.getElementsByTagName)(frame, "draw:image");
298
303
  if (drawImages.length > 0) {
299
304
  imageHref = drawImages[0].getAttribute("xlink:href") || '';
300
305
  if (imageHref) {
@@ -312,7 +317,7 @@ const parseOpenOffice = async (buffer, config) => {
312
317
  }
313
318
  };
314
319
  if (config.includeRawContent) {
315
- imageNode.rawContent = frame.toString();
320
+ imageNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frame, sourceXml, config);
316
321
  }
317
322
  children.push(imageNode);
318
323
  }
@@ -330,14 +335,14 @@ const parseOpenOffice = async (buffer, config) => {
330
335
  * @param config - Parser configuration
331
336
  * @returns Object containing text, children, alignment, and style info
332
337
  */
333
- const parseParagraphContent = (node, paraStyleMap, styleMap, config) => {
338
+ const parseParagraphContent = (node, paraStyleMap, styleMap, config, sourceXml) => {
334
339
  // Get paragraph style for alignment and drop caps
335
340
  const paraStyle = node.getAttribute("text:style-name");
336
341
  const styleInfo = paraStyle ? paraStyleMap[paraStyle] : undefined;
337
342
  const alignment = styleInfo?.alignment;
338
343
  const dropCap = styleInfo?.dropCap;
339
344
  // Parse content recursively using the new helper
340
- const content = parseInlineContent(node, styleMap, config, notes, paraStyleMap);
345
+ const content = parseInlineContent(node, styleMap, config, notes, paraStyleMap, {}, undefined, sourceXml);
341
346
  // Add style name to metadata of children if they don't have one
342
347
  if (paraStyle) {
343
348
  content.children.forEach(child => {
@@ -391,6 +396,31 @@ const parseOpenOffice = async (buffer, config) => {
391
396
  }
392
397
  return { text: content.text, children: content.children, alignment, style: paraStyle || undefined };
393
398
  };
399
+ /**
400
+ * Splits paragraph content into multiple segments based on line breaks.
401
+ * Used to handle soft line breaks within list items.
402
+ *
403
+ * @param pContent - The content of a single paragraph
404
+ * @returns Array of content segments
405
+ */
406
+ const splitParagraphByBreaks = (pContent) => {
407
+ const segments = [];
408
+ let currentText = "";
409
+ let currentChildren = [];
410
+ for (const child of pContent.children) {
411
+ if (child.type === "text" && child.metadata?.isLineBreak) {
412
+ segments.push({ text: currentText, children: currentChildren });
413
+ currentText = "";
414
+ currentChildren = [];
415
+ }
416
+ else {
417
+ currentText += child.text || "";
418
+ currentChildren.push(child);
419
+ }
420
+ }
421
+ segments.push({ text: currentText, children: currentChildren });
422
+ return segments;
423
+ };
394
424
  /**
395
425
  * Helper to parse a table node and extract its structure.
396
426
  * Properly creates table → row → cell hierarchy with metadata.
@@ -401,15 +431,15 @@ const parseOpenOffice = async (buffer, config) => {
401
431
  * @param config - Parser configuration
402
432
  * @returns Table content node with proper structure
403
433
  */
404
- const parseTable = (tableNode, paraStyleMap, styleMap, config) => {
434
+ const parseTable = (tableNode, paraStyleMap, styleMap, config, sourceXml) => {
405
435
  const rows = [];
406
436
  // Use getDirectChildren to avoid nested table rows
407
- const tableRows = (0, xmlUtils_1.getDirectChildren)(tableNode, "table:table-row");
437
+ const tableRows = (0, xmlUtils_js_1.getDirectChildren)(tableNode, "table:table-row");
408
438
  let rowIndex = 0;
409
439
  for (const row of tableRows) {
410
440
  const cells = [];
411
441
  // Use getDirectChildren to avoid nested table cells
412
- const tableCells = (0, xmlUtils_1.getDirectChildren)(row, "table:table-cell");
442
+ const tableCells = (0, xmlUtils_js_1.getDirectChildren)(row, "table:table-cell");
413
443
  const rowsRepeated = parseInt(row.getAttribute("table:number-rows-repeated") || "1");
414
444
  let colIndex = 0;
415
445
  for (const cell of tableCells) {
@@ -424,10 +454,10 @@ const parseOpenOffice = async (buffer, config) => {
424
454
  return;
425
455
  for (let i = 0; i < node.childNodes.length; i++) {
426
456
  const child = node.childNodes[i];
427
- if (child.nodeType === 1) { // Element
457
+ if ((0, xmlUtils_js_1.isElement)(child)) { // Element
428
458
  const element = child;
429
459
  if (element.tagName === "text:p" || element.tagName === "text:h") {
430
- const pContent = parseParagraphContent(element, paraStyleMap, styleMap, config);
460
+ const pContent = parseParagraphContent(element, paraStyleMap, styleMap, config, sourceXml);
431
461
  const pNode = {
432
462
  type: element.tagName === "text:h" ? 'heading' : 'paragraph',
433
463
  text: pContent.text,
@@ -446,7 +476,7 @@ const parseOpenOffice = async (buffer, config) => {
446
476
  pNode.metadata.level = parseInt(element.getAttribute("text:outline-level") || "1");
447
477
  }
448
478
  if (config.includeRawContent) {
449
- pNode.rawContent = element.toString();
479
+ pNode.rawContent = (0, xmlUtils_js_1.getRawContent)(element, sourceXml, config);
450
480
  }
451
481
  cellChildren.push(pNode);
452
482
  cellTextRef.value += pContent.text;
@@ -457,7 +487,7 @@ const parseOpenOffice = async (buffer, config) => {
457
487
  }
458
488
  else if (element.tagName === "table:table") {
459
489
  // Recursive call for nested table
460
- const nestedTableNode = parseTable(element, paraStyleMap, styleMap, config);
490
+ const nestedTableNode = parseTable(element, paraStyleMap, styleMap, config, sourceXml);
461
491
  cellChildren.push(nestedTableNode);
462
492
  }
463
493
  else if (element.tagName === "draw:frame" || element.tagName === "draw:text-box") {
@@ -487,7 +517,7 @@ const parseOpenOffice = async (buffer, config) => {
487
517
  if (rowSpan > 1)
488
518
  cellMetadata.rowSpan = rowSpan;
489
519
  if (config.includeRawContent) {
490
- cellNode.rawContent = cell.toString();
520
+ cellNode.rawContent = (0, xmlUtils_js_1.getRawContent)(cell, sourceXml, config);
491
521
  }
492
522
  cells.push(cellNode);
493
523
  colIndex++;
@@ -508,7 +538,7 @@ const parseOpenOffice = async (buffer, config) => {
508
538
  });
509
539
  }
510
540
  if (config.includeRawContent) {
511
- rowNode.rawContent = row.toString();
541
+ rowNode.rawContent = (0, xmlUtils_js_1.getRawContent)(row, sourceXml, config);
512
542
  }
513
543
  rows.push(rowNode);
514
544
  rowIndex++;
@@ -520,18 +550,20 @@ const parseOpenOffice = async (buffer, config) => {
520
550
  };
521
551
  };
522
552
  const parseContentXml = (xmlString) => {
523
- const xml = (0, xmlUtils_1.parseXmlString)(xmlString);
524
- const body = (0, xmlUtils_1.getElementsByTagName)(xml, "office:body")[0];
553
+ const xml = (0, xmlUtils_js_1.parseXmlString)(xmlString, { locator: config.includeRawContent });
554
+ const body = (0, xmlUtils_js_1.getFirstElementByTagName)(xml, "office:body");
555
+ if (!body)
556
+ return;
525
557
  // Parse automatic styles (local to content.xml)
526
- const automaticStyles = (0, xmlUtils_1.getElementsByTagName)(xml, "office:automatic-styles")[0];
558
+ const automaticStyles = (0, xmlUtils_js_1.getFirstElementByTagName)(xml, "office:automatic-styles");
527
559
  if (automaticStyles) {
528
- const styles = (0, xmlUtils_1.getElementsByTagName)(automaticStyles, "style:style");
560
+ const styles = (0, xmlUtils_js_1.getElementsByTagName)(automaticStyles, "style:style");
529
561
  for (const style of styles) {
530
562
  const name = style.getAttribute("style:name");
531
563
  if (!name)
532
564
  continue;
533
565
  // Parse paragraph properties for alignment
534
- const paraProps = (0, xmlUtils_1.getElementsByTagName)(style, "style:paragraph-properties")[0];
566
+ const paraProps = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "style:paragraph-properties");
535
567
  const styleInfo = {};
536
568
  if (paraProps) {
537
569
  const textAlign = paraProps.getAttribute("fo:text-align");
@@ -548,14 +580,14 @@ const parseOpenOffice = async (buffer, config) => {
548
580
  styleInfo.alignment = alignMap[textAlign];
549
581
  }
550
582
  }
551
- const dropCap = (0, xmlUtils_1.getElementsByTagName)(paraProps, "style:drop-cap")[0];
583
+ const dropCap = (0, xmlUtils_js_1.getFirstElementByTagName)(paraProps, "style:drop-cap");
552
584
  if (dropCap)
553
585
  styleInfo.dropCap = true;
554
586
  }
555
587
  if (Object.keys(styleInfo).length > 0) {
556
588
  paragraphStyleMap[name] = styleInfo;
557
589
  }
558
- const textProps = (0, xmlUtils_1.getElementsByTagName)(style, "style:text-properties")[0];
590
+ const textProps = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "style:text-properties");
559
591
  if (textProps) {
560
592
  const formatting = {};
561
593
  if (textProps.getAttribute("fo:font-weight") === "bold" || textProps.getAttribute("style:font-weight-asian") === "bold")
@@ -593,6 +625,19 @@ const parseOpenOffice = async (buffer, config) => {
593
625
  }
594
626
  }
595
627
  }
628
+ // Start traversal
629
+ const officeBody = (0, xmlUtils_js_1.getFirstElementByTagName)(xml, "office:body");
630
+ if (officeBody) {
631
+ const bodyContent = (0, xmlUtils_js_1.getDirectChildren)(officeBody, "office:text")[0] ||
632
+ (0, xmlUtils_js_1.getDirectChildren)(officeBody, "office:presentation")[0] ||
633
+ (0, xmlUtils_js_1.getDirectChildren)(officeBody, "office:spreadsheet")[0];
634
+ if (bodyContent) {
635
+ const bodyChildren = (0, xmlUtils_js_1.getDirectChildren)(bodyContent, "*");
636
+ for (const child of bodyChildren) {
637
+ traverse(child, content, false, xmlString);
638
+ }
639
+ }
640
+ }
596
641
  /**
597
642
  * Recursively traverses a node and its children to extract content.
598
643
  * Properly handles paragraphs, headings, tables, lists, and frames.
@@ -600,10 +645,11 @@ const parseOpenOffice = async (buffer, config) => {
600
645
  * @param node - The element to traverse
601
646
  * @param targetArray - The array to push extracted content nodes to
602
647
  * @param forceHeading - If true, treats all paragraphs as headings (used for slide titles)
648
+ * @param sourceXml - The source XML string for raw content extraction
603
649
  */
604
- const traverse = (node, targetArray, forceHeading = false) => {
650
+ function traverse(node, targetArray, forceHeading = false, sourceXml) {
605
651
  if (node.tagName === "text:p") {
606
- const pContent = parseParagraphContent(node, paragraphStyleMap, styleMap, config);
652
+ const pContent = parseParagraphContent(node, paragraphStyleMap, styleMap, config, sourceXml);
607
653
  const type = (forceHeading || (node.getAttribute("text:style-name") || '').toLowerCase().includes('title')) ? 'heading' : 'paragraph';
608
654
  const pNode = {
609
655
  type,
@@ -621,13 +667,14 @@ const parseOpenOffice = async (buffer, config) => {
621
667
  if (Object.keys(pNode.metadata || {}).length === 0)
622
668
  delete pNode.metadata;
623
669
  if (config.includeRawContent) {
624
- pNode.rawContent = node.toString();
670
+ pNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
625
671
  }
626
672
  targetArray.push(pNode);
673
+ lastWasList = false;
627
674
  }
628
675
  else if (node.tagName === "text:h") {
629
676
  const level = parseInt(node.getAttribute("text:outline-level") || "1");
630
- const hContent = parseParagraphContent(node, paragraphStyleMap, styleMap, config);
677
+ const hContent = parseParagraphContent(node, paragraphStyleMap, styleMap, config, sourceXml);
631
678
  const hNode = {
632
679
  type: 'heading',
633
680
  text: hContent.text,
@@ -639,27 +686,27 @@ const parseOpenOffice = async (buffer, config) => {
639
686
  }
640
687
  };
641
688
  if (config.includeRawContent) {
642
- hNode.rawContent = node.toString();
689
+ hNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
643
690
  }
644
691
  targetArray.push(hNode);
692
+ lastWasList = false;
645
693
  }
646
694
  else if (node.tagName === "table:table") {
647
695
  // Parse table with proper structure
648
- const tableNode = parseTable(node, paragraphStyleMap, styleMap, config);
696
+ const tableNode = parseTable(node, paragraphStyleMap, styleMap, config, sourceXml);
649
697
  if (config.includeRawContent) {
650
- tableNode.rawContent = node.toString();
698
+ tableNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
651
699
  }
652
700
  targetArray.push(tableNode);
701
+ lastWasList = false;
653
702
  }
654
703
  else if (node.tagName === "text:list") {
655
704
  // Parse list structure with proper listId tracking
656
- const listItems = (0, xmlUtils_1.getDirectChildren)(node, "text:list-item");
657
- // Get list style name to use as listId (or generate one)
658
- const listStyleName = node.getAttribute("text:style-name") || node.getAttribute("xml:id");
659
- const listId = listStyleName || `list-${targetArray.length}`;
705
+ const listItems = (0, xmlUtils_js_1.getDirectChildren)(node, "text:list-item");
660
706
  // Determine list type by checking the list style definition
661
707
  let listType = 'unordered';
662
708
  let isVisible = false;
709
+ const listStyleName = node.getAttribute("text:style-name") || node.getAttribute("xml:id");
663
710
  let styleNameToCheck = listStyleName;
664
711
  // If no style name, check parent list for inherited style
665
712
  if (!styleNameToCheck) {
@@ -675,15 +722,15 @@ const parseOpenOffice = async (buffer, config) => {
675
722
  }
676
723
  // Try to find list style in automatic styles to determine type and visibility
677
724
  if (styleNameToCheck) {
678
- const automaticStyles = (0, xmlUtils_1.getElementsByTagName)((0, xmlUtils_1.parseXmlString)(mainContentFile?.content.toString() || ''), "office:automatic-styles")[0];
725
+ const automaticStyles = (0, xmlUtils_js_1.getFirstElementByTagName)((0, xmlUtils_js_1.parseXmlString)(mainContentFile?.content.toString() || ''), "office:automatic-styles");
679
726
  if (automaticStyles) {
680
- const listStyles = (0, xmlUtils_1.getElementsByTagName)(automaticStyles, "text:list-style");
727
+ const listStyles = (0, xmlUtils_js_1.getElementsByTagName)(automaticStyles, "text:list-style");
681
728
  for (const listStyle of listStyles) {
682
729
  if (listStyle.getAttribute("style:name") === styleNameToCheck) {
683
730
  // Check if it has bullet or number level styles
684
- const bulletLevels = (0, xmlUtils_1.getElementsByTagName)(listStyle, "text:list-level-style-bullet");
685
- const numberLevels = (0, xmlUtils_1.getElementsByTagName)(listStyle, "text:list-level-style-number");
686
- const imageLevels = (0, xmlUtils_1.getElementsByTagName)(listStyle, "text:list-level-style-image");
731
+ const bulletLevels = (0, xmlUtils_js_1.getElementsByTagName)(listStyle, "text:list-level-style-bullet");
732
+ const numberLevels = (0, xmlUtils_js_1.getElementsByTagName)(listStyle, "text:list-level-style-number");
733
+ const imageLevels = (0, xmlUtils_js_1.getElementsByTagName)(listStyle, "text:list-level-style-image");
687
734
  if (numberLevels.length > 0) {
688
735
  listType = 'ordered';
689
736
  isVisible = numberLevels.some(l => !!l.getAttribute("style:num-format"));
@@ -700,13 +747,13 @@ const parseOpenOffice = async (buffer, config) => {
700
747
  }
701
748
  // Also check in styles.xml if still unordered and hidden
702
749
  if (stylesFile && !isVisible) {
703
- const stylesXml = (0, xmlUtils_1.parseXmlString)(stylesFile.content.toString());
704
- const listStyles = (0, xmlUtils_1.getElementsByTagName)(stylesXml, "text:list-style");
750
+ const stylesXml = (0, xmlUtils_js_1.parseXmlString)(stylesFile.content.toString());
751
+ const listStyles = (0, xmlUtils_js_1.getElementsByTagName)(stylesXml, "text:list-style");
705
752
  for (const listStyle of listStyles) {
706
753
  if (listStyle.getAttribute("style:name") === styleNameToCheck) {
707
- const bulletLevels = (0, xmlUtils_1.getElementsByTagName)(listStyle, "text:list-level-style-bullet");
708
- const numberLevels = (0, xmlUtils_1.getElementsByTagName)(listStyle, "text:list-level-style-number");
709
- const imageLevels = (0, xmlUtils_1.getElementsByTagName)(listStyle, "text:list-level-style-image");
754
+ const bulletLevels = (0, xmlUtils_js_1.getElementsByTagName)(listStyle, "text:list-level-style-bullet");
755
+ const numberLevels = (0, xmlUtils_js_1.getElementsByTagName)(listStyle, "text:list-level-style-number");
756
+ const imageLevels = (0, xmlUtils_js_1.getElementsByTagName)(listStyle, "text:list-level-style-image");
710
757
  if (numberLevels.length > 0) {
711
758
  listType = 'ordered';
712
759
  isVisible = numberLevels.some(l => !!l.getAttribute("style:num-format"));
@@ -725,19 +772,38 @@ const parseOpenOffice = async (buffer, config) => {
725
772
  // If the list is not visible, it's likely a layout list used by Impress.
726
773
  // We should traverse its items and treat their content as regular nodes.
727
774
  if (!isVisible) {
775
+ lastWasList = false;
728
776
  for (let i = 0; i < listItems.length; i++) {
729
777
  const item = listItems[i];
730
778
  if (item.childNodes) {
731
779
  for (let j = 0; j < item.childNodes.length; j++) {
732
780
  const child = item.childNodes[j];
733
- if (child.nodeType === 1) { // Element
734
- traverse(child, targetArray, forceHeading);
781
+ if ((0, xmlUtils_js_1.isElement)(child)) { // Element
782
+ traverse(child, targetArray, forceHeading, sourceXml);
735
783
  }
736
784
  }
737
785
  }
738
786
  }
739
787
  return;
740
788
  }
789
+ // List Continuity Logic:
790
+ // If this list follows another list of the same type and style, or we are in ODP and it's sequential,
791
+ // we should reuse the previous listId to maintain numbering.
792
+ const isODP = fileType === 'odp';
793
+ const sameStyle = styleNameToCheck && styleNameToCheck === lastListStyle;
794
+ const sameType = listType === lastListType;
795
+ let listId;
796
+ if (lastWasList && (sameStyle || (isODP && sameType))) {
797
+ listId = currentListId;
798
+ }
799
+ else {
800
+ // New list
801
+ listId = styleNameToCheck || `list-${++listIdCounter}`;
802
+ currentListId = listId;
803
+ lastListType = listType;
804
+ lastListStyle = styleNameToCheck;
805
+ }
806
+ lastWasList = true;
741
807
  // Calculate indentation level by counting parent text:list elements
742
808
  let indentation = 0;
743
809
  let parent = node.parentNode;
@@ -758,63 +824,61 @@ const parseOpenOffice = async (buffer, config) => {
758
824
  // Process each list item
759
825
  for (let i = 0; i < listItems.length; i++) {
760
826
  const item = listItems[i];
761
- // Increment item index for this list/level
762
- listCounters[listId][indentKey]++;
763
- const itemIndex = listCounters[listId][indentKey];
764
- // Reset deeper levels when we encounter an item at this level
765
- for (let k = indentation + 1; k < 10; k++) {
766
- if (listCounters[listId][k.toString()] !== undefined) {
767
- listCounters[listId][k.toString()] = -1;
768
- }
769
- }
827
+ let hasIndexedThisItem = false;
770
828
  // Iterate over direct children of list item (paragraphs, headings, nested lists)
771
829
  if (item.childNodes) {
772
830
  for (let j = 0; j < item.childNodes.length; j++) {
773
831
  const child = item.childNodes[j];
774
- if (child.nodeType === 1) { // Element
832
+ if ((0, xmlUtils_js_1.isElement)(child)) { // Element
775
833
  const element = child;
776
- if (element.tagName === "text:p") {
777
- const pContent = parseParagraphContent(element, paragraphStyleMap, styleMap, config);
778
- const listNode = {
779
- type: 'list',
780
- text: pContent.text,
781
- children: pContent.children,
782
- metadata: {
783
- listType,
784
- indentation,
785
- itemIndex,
786
- listId,
787
- alignment: pContent.alignment || 'left',
788
- style: pContent.style
834
+ if (element.tagName === "text:p" || element.tagName === "text:h") {
835
+ if (!hasIndexedThisItem) {
836
+ listCounters[listId][indentKey]++;
837
+ hasIndexedThisItem = true;
838
+ for (let k = indentation + 1; k < 10; k++) {
839
+ if (listCounters[listId][k.toString()] !== undefined) {
840
+ listCounters[listId][k.toString()] = -1;
841
+ }
789
842
  }
790
- };
791
- if (config.includeRawContent)
792
- listNode.rawContent = element.toString();
793
- targetArray.push(listNode);
794
- }
795
- else if (element.tagName === "text:h") {
796
- const level = parseInt(element.getAttribute("text:outline-level") || "1");
797
- const hContent = parseParagraphContent(element, paragraphStyleMap, styleMap, config);
798
- const listNode = {
799
- type: 'list',
800
- text: hContent.text,
801
- children: hContent.children,
802
- metadata: {
803
- listType,
804
- indentation,
805
- itemIndex,
806
- listId,
807
- ...(hContent.alignment ? { alignment: hContent.alignment } : {}),
808
- style: hContent.style
843
+ }
844
+ const itemIndex = listCounters[listId][indentKey];
845
+ const pContent = parseParagraphContent(element, paragraphStyleMap, styleMap, config, sourceXml);
846
+ const segments = splitParagraphByBreaks(pContent);
847
+ for (let k = 0; k < segments.length; k++) {
848
+ const segment = segments[k];
849
+ if (!segment.text.trim() && segment.children.length === 0)
850
+ continue;
851
+ const isFirst = k === 0;
852
+ const nodeType = isFirst ? 'list' : 'paragraph';
853
+ const node = {
854
+ type: nodeType,
855
+ text: segment.text,
856
+ children: segment.children,
857
+ metadata: isFirst ? {
858
+ listType,
859
+ indentation,
860
+ itemIndex,
861
+ listId,
862
+ alignment: pContent.alignment || 'left',
863
+ style: pContent.style
864
+ } : {
865
+ alignment: pContent.alignment || 'left',
866
+ style: pContent.style
867
+ }
868
+ };
869
+ // Special case for headings in lists
870
+ if (isFirst && element.tagName === "text:h") {
871
+ const level = parseInt(element.getAttribute("text:outline-level") || "1");
872
+ node.metadata.level = level;
809
873
  }
810
- };
811
- if (config.includeRawContent)
812
- listNode.rawContent = element.toString();
813
- targetArray.push(listNode);
874
+ if (config.includeRawContent)
875
+ node.rawContent = (0, xmlUtils_js_1.getRawContent)(element, sourceXml, config);
876
+ targetArray.push(node);
877
+ }
814
878
  }
815
879
  else if (element.tagName === "text:list") {
816
880
  // Recursive call for nested list
817
- traverse(element, targetArray, forceHeading);
881
+ traverse(element, targetArray, forceHeading, sourceXml);
818
882
  }
819
883
  }
820
884
  }
@@ -825,24 +889,24 @@ const parseOpenOffice = async (buffer, config) => {
825
889
  const presClass = node.getAttribute("presentation:class");
826
890
  const isHeading = presClass === "title" || presClass === "sub-title";
827
891
  // In presentations, frames often contain text-boxes, images, tables, or objects
828
- const textBox = (0, xmlUtils_1.getElementsByTagName)(node, "draw:text-box")[0];
829
- const image = (0, xmlUtils_1.getElementsByTagName)(node, "draw:image")[0];
830
- const table = (0, xmlUtils_1.getElementsByTagName)(node, "table:table")[0];
831
- const object = (0, xmlUtils_1.getElementsByTagName)(node, "draw:object")[0];
892
+ const textBox = (0, xmlUtils_js_1.getFirstElementByTagName)(node, "draw:text-box");
893
+ const image = (0, xmlUtils_js_1.getFirstElementByTagName)(node, "draw:image");
894
+ const table = (0, xmlUtils_js_1.getFirstElementByTagName)(node, "table:table");
895
+ const object = (0, xmlUtils_js_1.getFirstElementByTagName)(node, "draw:object");
832
896
  if (textBox) {
833
- traverse(textBox, targetArray, isHeading || forceHeading);
897
+ traverse(textBox, targetArray, isHeading || forceHeading, sourceXml);
834
898
  }
835
899
  else if (table) {
836
- const tableNode = parseTable(table, paragraphStyleMap, styleMap, config);
900
+ const tableNode = parseTable(table, paragraphStyleMap, styleMap, config, sourceXml);
837
901
  if (config.includeRawContent)
838
- tableNode.rawContent = table.toString();
902
+ tableNode.rawContent = (0, xmlUtils_js_1.getRawContent)(table, sourceXml, config);
839
903
  targetArray.push(tableNode);
840
904
  }
841
905
  else if (image) {
842
906
  // Extract alt text from svg:title or svg:desc
843
907
  let altText = '';
844
- const svgTitle = (0, xmlUtils_1.getElementsByTagName)(node, "svg:title")[0];
845
- const svgDesc = (0, xmlUtils_1.getElementsByTagName)(node, "svg:desc")[0];
908
+ const svgTitle = (0, xmlUtils_js_1.getFirstElementByTagName)(node, "svg:title");
909
+ const svgDesc = (0, xmlUtils_js_1.getFirstElementByTagName)(node, "svg:desc");
846
910
  if (svgTitle && svgTitle.textContent) {
847
911
  altText = svgTitle.textContent;
848
912
  }
@@ -865,7 +929,7 @@ const parseOpenOffice = async (buffer, config) => {
865
929
  }
866
930
  };
867
931
  if (config.includeRawContent) {
868
- imageNode.rawContent = node.toString();
932
+ imageNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
869
933
  }
870
934
  targetArray.push(imageNode);
871
935
  }
@@ -877,7 +941,7 @@ const parseOpenOffice = async (buffer, config) => {
877
941
  const objectPath = `${attachmentName}/content.xml`;
878
942
  const objectFile = files.find(f => f.path === objectPath || f.path.endsWith(objectPath));
879
943
  if (objectFile) {
880
- const chartData = (0, chartUtils_1.extractChartData)(objectFile.content);
944
+ const chartData = (0, chartUtils_js_1.extractChartData)(objectFile.content);
881
945
  const chartNode = {
882
946
  type: 'chart',
883
947
  text: chartData.rawTexts.join(" "),
@@ -887,7 +951,7 @@ const parseOpenOffice = async (buffer, config) => {
887
951
  }
888
952
  };
889
953
  if (config.includeRawContent)
890
- chartNode.rawContent = node.toString();
954
+ chartNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
891
955
  targetArray.push(chartNode);
892
956
  }
893
957
  else {
@@ -897,7 +961,7 @@ const parseOpenOffice = async (buffer, config) => {
897
961
  metadata: { attachmentName: attachmentName }
898
962
  };
899
963
  if (config.includeRawContent)
900
- chartNode.rawContent = node.toString();
964
+ chartNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
901
965
  targetArray.push(chartNode);
902
966
  }
903
967
  }
@@ -907,28 +971,29 @@ const parseOpenOffice = async (buffer, config) => {
907
971
  if (node.childNodes) {
908
972
  for (let i = 0; i < node.childNodes.length; i++) {
909
973
  const child = node.childNodes[i];
910
- if (child.nodeType === 1) { // Element
911
- traverse(child, targetArray, forceHeading);
974
+ if ((0, xmlUtils_js_1.isElement)(child)) { // Element
975
+ traverse(child, targetArray, forceHeading, sourceXml);
912
976
  }
913
977
  }
914
978
  }
915
979
  }
916
- };
980
+ }
981
+ ;
917
982
  // ODS: Spreadsheet
918
983
  if (fileType === 'ods') {
919
- const spreadsheet = (0, xmlUtils_1.getElementsByTagName)(body, "office:spreadsheet")[0];
984
+ const spreadsheet = (0, xmlUtils_js_1.getFirstElementByTagName)(body, "office:spreadsheet");
920
985
  if (spreadsheet) {
921
- const tables = (0, xmlUtils_1.getElementsByTagName)(spreadsheet, "table:table");
986
+ const tables = (0, xmlUtils_js_1.getElementsByTagName)(spreadsheet, "table:table");
922
987
  for (let i = 0; i < tables.length; i++) {
923
988
  const table = tables[i];
924
989
  const sheetName = table.getAttribute("table:name") || `Sheet${i + 1}`;
925
990
  const rows = [];
926
- const tableRows = (0, xmlUtils_1.getElementsByTagName)(table, "table:table-row");
991
+ const tableRows = (0, xmlUtils_js_1.getElementsByTagName)(table, "table:table-row");
927
992
  let rowIndex = 0;
928
993
  for (let r = 0; r < tableRows.length; r++) {
929
994
  const row = tableRows[r];
930
995
  const cells = [];
931
- const tableCells = (0, xmlUtils_1.getElementsByTagName)(row, "table:table-cell");
996
+ const tableCells = (0, xmlUtils_js_1.getElementsByTagName)(row, "table:table-cell");
932
997
  let colIndex = 0;
933
998
  const rowsRepeated = parseInt(row.getAttribute("table:number-rows-repeated") || "1");
934
999
  for (let c = 0; c < tableCells.length; c++) {
@@ -937,11 +1002,11 @@ const parseOpenOffice = async (buffer, config) => {
937
1002
  // Extract text from cell (paragraphs inside cell)
938
1003
  let cellText = "";
939
1004
  const children = [];
940
- const ps = (0, xmlUtils_1.getElementsByTagName)(cell, "text:p");
1005
+ const ps = (0, xmlUtils_js_1.getElementsByTagName)(cell, "text:p");
941
1006
  for (let p = 0; p < ps.length; p++) {
942
1007
  const para = ps[p];
943
1008
  // Parse text:span elements for formatted text
944
- const spans = (0, xmlUtils_1.getElementsByTagName)(para, "text:span");
1009
+ const spans = (0, xmlUtils_js_1.getElementsByTagName)(para, "text:span");
945
1010
  if (spans.length > 0) {
946
1011
  for (const span of spans) {
947
1012
  const styleName = span.getAttribute("text:style-name");
@@ -973,12 +1038,12 @@ const parseOpenOffice = async (buffer, config) => {
973
1038
  cellText += "\n";
974
1039
  }
975
1040
  // Check for embedded draw:frame (images) in cell
976
- const drawFrames = (0, xmlUtils_1.getElementsByTagName)(cell, "draw:frame");
1041
+ const drawFrames = (0, xmlUtils_js_1.getElementsByTagName)(cell, "draw:frame");
977
1042
  for (const frame of drawFrames) {
978
1043
  // Extract alt text from svg:title or svg:desc
979
1044
  let altText = '';
980
- const svgTitle = (0, xmlUtils_1.getElementsByTagName)(frame, "svg:title")[0];
981
- const svgDesc = (0, xmlUtils_1.getElementsByTagName)(frame, "svg:desc")[0];
1045
+ const svgTitle = (0, xmlUtils_js_1.getFirstElementByTagName)(frame, "svg:title");
1046
+ const svgDesc = (0, xmlUtils_js_1.getFirstElementByTagName)(frame, "svg:desc");
982
1047
  if (svgTitle && svgTitle.textContent) {
983
1048
  altText = svgTitle.textContent;
984
1049
  }
@@ -987,7 +1052,7 @@ const parseOpenOffice = async (buffer, config) => {
987
1052
  }
988
1053
  // Extract image href
989
1054
  let imageHref = '';
990
- const drawImages = (0, xmlUtils_1.getElementsByTagName)(frame, "draw:image");
1055
+ const drawImages = (0, xmlUtils_js_1.getElementsByTagName)(frame, "draw:image");
991
1056
  if (drawImages.length > 0) {
992
1057
  const rawHref = drawImages[0].getAttribute("xlink:href");
993
1058
  if (rawHref) {
@@ -997,7 +1062,7 @@ const parseOpenOffice = async (buffer, config) => {
997
1062
  }
998
1063
  // Extract chart object href
999
1064
  let chartHref = '';
1000
- const drawObjects = (0, xmlUtils_1.getElementsByTagName)(frame, "draw:object");
1065
+ const drawObjects = (0, xmlUtils_js_1.getElementsByTagName)(frame, "draw:object");
1001
1066
  if (drawObjects.length > 0) {
1002
1067
  const href = drawObjects[0].getAttribute("xlink:href");
1003
1068
  if (href) {
@@ -1017,7 +1082,7 @@ const parseOpenOffice = async (buffer, config) => {
1017
1082
  }
1018
1083
  };
1019
1084
  if (config.includeRawContent) {
1020
- imageNode.rawContent = frame.toString();
1085
+ imageNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frame, xmlString, config);
1021
1086
  }
1022
1087
  children.push(imageNode);
1023
1088
  }
@@ -1046,7 +1111,7 @@ const parseOpenOffice = async (buffer, config) => {
1046
1111
  metadata: { row: rowIndex, col: colIndex }
1047
1112
  };
1048
1113
  if (config.includeRawContent) {
1049
- cellNode.rawContent = cell.toString();
1114
+ cellNode.rawContent = (0, xmlUtils_js_1.getRawContent)(cell, xmlString, config);
1050
1115
  }
1051
1116
  cells.push(cellNode);
1052
1117
  }
@@ -1070,7 +1135,7 @@ const parseOpenOffice = async (buffer, config) => {
1070
1135
  });
1071
1136
  }
1072
1137
  if (config.includeRawContent) {
1073
- rowNode.rawContent = row.toString();
1138
+ rowNode.rawContent = (0, xmlUtils_js_1.getRawContent)(row, xmlString, config);
1074
1139
  }
1075
1140
  rows.push(rowNode);
1076
1141
  rowIndex++;
@@ -1086,7 +1151,7 @@ const parseOpenOffice = async (buffer, config) => {
1086
1151
  metadata: { sheetName }
1087
1152
  };
1088
1153
  if (config.includeRawContent) {
1089
- sheetNode.rawContent = table.toString();
1154
+ sheetNode.rawContent = (0, xmlUtils_js_1.getRawContent)(table, xmlString, config);
1090
1155
  }
1091
1156
  content.push(sheetNode);
1092
1157
  }
@@ -1094,9 +1159,9 @@ const parseOpenOffice = async (buffer, config) => {
1094
1159
  }
1095
1160
  // ODP: Presentation
1096
1161
  else if (fileType === 'odp') {
1097
- const presentation = (0, xmlUtils_1.getElementsByTagName)(body, "office:presentation")[0];
1162
+ const presentation = (0, xmlUtils_js_1.getFirstElementByTagName)(body, "office:presentation");
1098
1163
  if (presentation) {
1099
- const pages = (0, xmlUtils_1.getDirectChildren)(presentation, "draw:page");
1164
+ const pages = (0, xmlUtils_js_1.getDirectChildren)(presentation, "draw:page");
1100
1165
  const odpNotes = [];
1101
1166
  for (let i = 0; i < pages.length; i++) {
1102
1167
  const page = pages[i];
@@ -1111,7 +1176,7 @@ const parseOpenOffice = async (buffer, config) => {
1111
1176
  if (pageChildren) {
1112
1177
  for (let j = 0; j < pageChildren.length; j++) {
1113
1178
  const child = pageChildren[j];
1114
- if (child.nodeType === 1) { // Element
1179
+ if ((0, xmlUtils_js_1.isElement)(child)) { // Element
1115
1180
  const element = child;
1116
1181
  if (element.tagName === "presentation:notes") {
1117
1182
  if (!config.ignoreNotes) {
@@ -1123,16 +1188,16 @@ const parseOpenOffice = async (buffer, config) => {
1123
1188
  noteId: `slide-note-${i + 1}`
1124
1189
  }
1125
1190
  };
1126
- traverse(element, noteNode.children);
1191
+ traverse(element, noteNode.children, false, xmlString);
1127
1192
  }
1128
1193
  continue;
1129
1194
  }
1130
- traverse(element, slideNode.children);
1195
+ traverse(element, slideNode.children, false, xmlString);
1131
1196
  }
1132
1197
  }
1133
1198
  }
1134
1199
  if (config.includeRawContent) {
1135
- slideNode.rawContent = page.toString();
1200
+ slideNode.rawContent = (0, xmlUtils_js_1.getRawContent)(page, xmlString, config);
1136
1201
  }
1137
1202
  content.push(slideNode);
1138
1203
  if (noteNode && noteNode.children && noteNode.children.length > 0) {
@@ -1151,9 +1216,9 @@ const parseOpenOffice = async (buffer, config) => {
1151
1216
  }
1152
1217
  // ODT: Text Document (and generic fallback)
1153
1218
  else {
1154
- const textDoc = (0, xmlUtils_1.getElementsByTagName)(body, "office:text")[0];
1219
+ const textDoc = (0, xmlUtils_js_1.getFirstElementByTagName)(body, "office:text");
1155
1220
  if (textDoc) {
1156
- traverse(textDoc, content);
1221
+ traverse(textDoc, content, false, xmlString);
1157
1222
  }
1158
1223
  }
1159
1224
  };
@@ -1167,8 +1232,8 @@ const parseOpenOffice = async (buffer, config) => {
1167
1232
  if (config.extractAttachments) {
1168
1233
  const objectFiles = files.filter(f => f.path.match(/Object \d+\/content\.xml/));
1169
1234
  for (const objFile of objectFiles) {
1170
- const objXml = (0, xmlUtils_1.parseXmlString)(objFile.content.toString());
1171
- const isChart = (0, xmlUtils_1.getElementsByTagName)(objXml, "chart:chart").length > 0;
1235
+ const objXml = (0, xmlUtils_js_1.parseXmlString)(objFile.content.toString());
1236
+ const isChart = (0, xmlUtils_js_1.getElementsByTagName)(objXml, "chart:chart").length > 0;
1172
1237
  if (isChart) {
1173
1238
  const objectId = objFile.path.split('/')[0];
1174
1239
  const attachment = {
@@ -1179,7 +1244,7 @@ const parseOpenOffice = async (buffer, config) => {
1179
1244
  extension: 'xml'
1180
1245
  };
1181
1246
  // Extract data from chart XML
1182
- const chartData = (0, chartUtils_1.extractChartData)(objFile.content);
1247
+ const chartData = (0, chartUtils_js_1.extractChartData)(objFile.content);
1183
1248
  if (chartData.rawTexts.length > 0) {
1184
1249
  attachment.chartData = chartData;
1185
1250
  }
@@ -1189,22 +1254,22 @@ const parseOpenOffice = async (buffer, config) => {
1189
1254
  }
1190
1255
  if (config.extractAttachments) {
1191
1256
  for (const media of mediaFiles) {
1192
- const attachment = (0, imageUtils_1.createAttachment)(media.path.split('/').pop() || 'image', media.content);
1257
+ const attachment = (0, imageUtils_js_1.createAttachment)(media.path.split('/').pop() || 'image', media.content);
1193
1258
  attachments.push(attachment);
1194
1259
  if (config.ocr) {
1195
1260
  if (attachment.mimeType.startsWith('image/')) {
1196
1261
  try {
1197
- attachment.ocrText = (await (0, ocrUtils_1.performOcr)(media.content, config.ocrLanguage)).trim();
1262
+ attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(media.content, { language: config.ocrLanguage, ...config.ocrConfig })).trim();
1198
1263
  }
1199
1264
  catch (e) {
1200
- (0, errorUtils_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
1265
+ (0, errorUtils_js_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
1201
1266
  }
1202
1267
  }
1203
1268
  }
1204
1269
  }
1205
1270
  }
1206
1271
  const metaFile = files.find(f => f.path.match(metaFileRegex));
1207
- const metadata = metaFile ? (0, xmlUtils_1.parseOfficeMetadata)(metaFile.content.toString()) : {};
1272
+ const metadata = metaFile ? (0, xmlUtils_js_1.parseOfficeMetadata)(metaFile.content.toString()) : {};
1208
1273
  // Helper: Resolve ODS chart cell references to actual values
1209
1274
  // ODS charts often link to cell ranges (e.g., [Sheet1.$A$1:.$A$5]) instead of embedding values
1210
1275
  const resolveChartReferences = (chartData, nodes) => {