officeparser 6.0.7 → 6.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/README.md +92 -13
  2. package/dist/OfficeParser.d.ts +10 -1
  3. package/dist/OfficeParser.js +43 -56
  4. package/dist/cli.d.ts +20 -0
  5. package/dist/cli.js +116 -0
  6. package/dist/index.d.ts +3 -3
  7. package/dist/index.js +7 -59
  8. package/dist/index.mjs +18 -0
  9. package/dist/officeparser.browser.d.ts +79 -1
  10. package/dist/officeparser.browser.iife.js +112 -0
  11. package/dist/officeparser.browser.mjs +111 -0
  12. package/dist/parsers/ExcelParser.d.ts +1 -1
  13. package/dist/parsers/ExcelParser.js +71 -63
  14. package/dist/parsers/OpenOfficeParser.d.ts +1 -1
  15. package/dist/parsers/OpenOfficeParser.js +131 -114
  16. package/dist/parsers/PdfParser.d.ts +1 -1
  17. package/dist/parsers/PdfParser.js +98 -94
  18. package/dist/parsers/PowerPointParser.d.ts +1 -1
  19. package/dist/parsers/PowerPointParser.js +85 -88
  20. package/dist/parsers/RtfParser.d.ts +1 -1
  21. package/dist/parsers/RtfParser.js +10 -6
  22. package/dist/parsers/WordParser.d.ts +1 -1
  23. package/dist/parsers/WordParser.js +109 -101
  24. package/dist/sbom.cdx.json +1807 -0
  25. package/dist/types.d.ts +69 -1
  26. package/dist/utils/chartUtils.js +2 -0
  27. package/dist/utils/dateUtils.d.ts +17 -0
  28. package/dist/utils/dateUtils.js +69 -0
  29. package/dist/utils/envUtils.d.ts +24 -0
  30. package/dist/utils/envUtils.js +69 -0
  31. package/dist/utils/moduleLoader.d.ts +2 -1
  32. package/dist/utils/moduleLoader.js +9 -39
  33. package/dist/utils/ocrUtils.d.ts +16 -12
  34. package/dist/utils/ocrUtils.js +186 -25
  35. package/dist/utils/xmlUtils.d.ts +80 -9
  36. package/dist/utils/xmlUtils.js +236 -18
  37. package/dist/utils/zipUtils.js +6 -47
  38. package/package.json +31 -16
  39. package/dist/officeParserBundle@6.0.7.js +0 -154
  40. package/dist/officeparser.browser.js +0 -154
@@ -23,12 +23,12 @@
23
23
  */
24
24
  Object.defineProperty(exports, "__esModule", { value: true });
25
25
  exports.parseOpenOffice = void 0;
26
- const chartUtils_1 = require("../utils/chartUtils");
27
- const errorUtils_1 = require("../utils/errorUtils");
28
- const imageUtils_1 = require("../utils/imageUtils");
29
- const ocrUtils_1 = require("../utils/ocrUtils");
30
- const xmlUtils_1 = require("../utils/xmlUtils");
31
- const zipUtils_1 = require("../utils/zipUtils");
26
+ const chartUtils_js_1 = require("../utils/chartUtils.js");
27
+ const errorUtils_js_1 = require("../utils/errorUtils.js");
28
+ const imageUtils_js_1 = require("../utils/imageUtils.js");
29
+ const ocrUtils_js_1 = require("../utils/ocrUtils.js");
30
+ const xmlUtils_js_1 = require("../utils/xmlUtils.js");
31
+ const zipUtils_js_1 = require("../utils/zipUtils.js");
32
32
  /**
33
33
  * Parses an OpenOffice document (.odt, .odp, .ods) and extracts content.
34
34
  *
@@ -43,7 +43,7 @@ const parseOpenOffice = async (buffer, config) => {
43
43
  const metaFileRegex = /meta\.xml/;
44
44
  const stylesFileRegex = /styles\.xml/;
45
45
  const mimetypeFileRegex = /mimetype/;
46
- const files = await (0, zipUtils_1.extractFiles)(buffer, x => !!x.match(contentFileRegex) ||
46
+ const files = await (0, zipUtils_js_1.extractFiles)(buffer, x => !!x.match(contentFileRegex) ||
47
47
  !!x.match(objectContentFileRegex) ||
48
48
  !!x.match(metaFileRegex) ||
49
49
  !!x.match(stylesFileRegex) ||
@@ -72,15 +72,15 @@ const parseOpenOffice = async (buffer, config) => {
72
72
  const listCounters = {}; // Track item index per listId/level
73
73
  // Helper to parse styles
74
74
  const parseStyles = (xmlString) => {
75
- const xml = (0, xmlUtils_1.parseXmlString)(xmlString);
76
- const styles = (0, xmlUtils_1.getElementsByTagName)(xml, "style:style");
75
+ const xml = (0, xmlUtils_js_1.parseXmlString)(xmlString);
76
+ const styles = (0, xmlUtils_js_1.getElementsByTagName)(xml, "style:style");
77
77
  for (const style of styles) {
78
78
  const name = style.getAttribute("style:name");
79
79
  if (!name)
80
80
  continue;
81
81
  const styleInfo = {};
82
82
  // Parse paragraph properties for alignment and drop caps
83
- const paraProps = (0, xmlUtils_1.getElementsByTagName)(style, "style:paragraph-properties")[0];
83
+ const paraProps = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "style:paragraph-properties");
84
84
  if (paraProps) {
85
85
  const textAlign = paraProps.getAttribute("fo:text-align");
86
86
  if (textAlign) {
@@ -97,7 +97,7 @@ const parseOpenOffice = async (buffer, config) => {
97
97
  }
98
98
  }
99
99
  // Detect Drop Caps
100
- const dropCap = (0, xmlUtils_1.getElementsByTagName)(paraProps, "style:drop-cap")[0];
100
+ const dropCap = (0, xmlUtils_js_1.getFirstElementByTagName)(paraProps, "style:drop-cap");
101
101
  if (dropCap) {
102
102
  styleInfo.dropCap = true;
103
103
  }
@@ -106,9 +106,9 @@ const parseOpenOffice = async (buffer, config) => {
106
106
  paragraphStyleMap[name] = styleInfo;
107
107
  }
108
108
  // Parse text properties
109
- const textProps = (0, xmlUtils_1.getElementsByTagName)(style, "style:text-properties")[0];
109
+ const textProps = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "style:text-properties");
110
110
  // Parse table cell properties (for ODS background)
111
- const cellProps = (0, xmlUtils_1.getElementsByTagName)(style, "style:table-cell-properties")[0];
111
+ const cellProps = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "style:table-cell-properties");
112
112
  const formatting = {};
113
113
  if (cellProps) {
114
114
  const bgColor = cellProps.getAttribute("fo:background-color");
@@ -170,7 +170,7 @@ const parseOpenOffice = async (buffer, config) => {
170
170
  * @param linkMetadata - Metadata inherited from parent link
171
171
  * @returns Object containing text and children
172
172
  */
173
- const parseInlineContent = (node, styleMap, config, notes, paragraphStyleMap, parentFormatting = {}, linkMetadata) => {
173
+ const parseInlineContent = (node, styleMap, config, notes, paragraphStyleMap, parentFormatting = {}, linkMetadata, sourceXml = '') => {
174
174
  const children = [];
175
175
  let fullText = '';
176
176
  if (!node.childNodes)
@@ -189,7 +189,7 @@ const parseOpenOffice = async (buffer, config) => {
189
189
  });
190
190
  }
191
191
  }
192
- else if (child.nodeType === 1) {
192
+ else if ((0, xmlUtils_js_1.isElement)(child)) {
193
193
  const element = child;
194
194
  const tagName = element.tagName;
195
195
  if (tagName === 'text:s') {
@@ -228,7 +228,7 @@ const parseOpenOffice = async (buffer, config) => {
228
228
  // Formatted text span
229
229
  const styleName = element.getAttribute("text:style-name");
230
230
  const formatting = styleName ? { ...parentFormatting, ...styleMap[styleName] } : parentFormatting;
231
- const spanContent = parseInlineContent(element, styleMap, config, notes, paragraphStyleMap, formatting, linkMetadata);
231
+ const spanContent = parseInlineContent(element, styleMap, config, notes, paragraphStyleMap, formatting, linkMetadata, sourceXml);
232
232
  fullText += spanContent.text;
233
233
  children.push(...spanContent.children);
234
234
  }
@@ -237,7 +237,7 @@ const parseOpenOffice = async (buffer, config) => {
237
237
  const href = element.getAttribute('xlink:href') || '';
238
238
  const linkType = href.startsWith('#') ? 'internal' : 'external';
239
239
  const newLinkMetadata = { link: href, linkType: linkType };
240
- const linkContent = parseInlineContent(element, styleMap, config, notes, paragraphStyleMap, parentFormatting, newLinkMetadata);
240
+ const linkContent = parseInlineContent(element, styleMap, config, notes, paragraphStyleMap, parentFormatting, newLinkMetadata, sourceXml);
241
241
  fullText += linkContent.text;
242
242
  children.push(...linkContent.children);
243
243
  }
@@ -245,14 +245,14 @@ const parseOpenOffice = async (buffer, config) => {
245
245
  // Footnote or endnote
246
246
  const noteClass = (element.getAttribute('text:note-class') || 'footnote');
247
247
  const noteId = element.getAttribute('text:id') || element.getAttribute('xml:id') || undefined;
248
- const noteBody = (0, xmlUtils_1.getElementsByTagName)(element, "text:note-body")[0];
248
+ const noteBody = (0, xmlUtils_js_1.getFirstElementByTagName)(element, "text:note-body");
249
249
  if (noteBody) {
250
250
  // Extract note content recursively
251
- const notePs = (0, xmlUtils_1.getElementsByTagName)(noteBody, "text:p");
251
+ const notePs = (0, xmlUtils_js_1.getElementsByTagName)(noteBody, "text:p");
252
252
  const noteChildren = [];
253
253
  let noteText = '';
254
254
  for (const np of notePs) {
255
- const npContent = parseParagraphContent(np, paragraphStyleMap, styleMap, config);
255
+ const npContent = parseParagraphContent(np, paragraphStyleMap, styleMap, config, sourceXml);
256
256
  noteText += (noteText ? ' ' : '') + npContent.text;
257
257
  const npNode = {
258
258
  type: 'paragraph',
@@ -284,8 +284,8 @@ const parseOpenOffice = async (buffer, config) => {
284
284
  const frame = element;
285
285
  // Extract alt text
286
286
  let altText = '';
287
- const svgTitle = (0, xmlUtils_1.getElementsByTagName)(frame, "svg:title")[0];
288
- const svgDesc = (0, xmlUtils_1.getElementsByTagName)(frame, "svg:desc")[0];
287
+ const svgTitle = (0, xmlUtils_js_1.getFirstElementByTagName)(frame, "svg:title");
288
+ const svgDesc = (0, xmlUtils_js_1.getFirstElementByTagName)(frame, "svg:desc");
289
289
  if (svgTitle && svgTitle.textContent) {
290
290
  altText = svgTitle.textContent;
291
291
  }
@@ -294,7 +294,7 @@ const parseOpenOffice = async (buffer, config) => {
294
294
  }
295
295
  // Extract image href
296
296
  let imageHref = '';
297
- const drawImages = (0, xmlUtils_1.getElementsByTagName)(frame, "draw:image");
297
+ const drawImages = (0, xmlUtils_js_1.getElementsByTagName)(frame, "draw:image");
298
298
  if (drawImages.length > 0) {
299
299
  imageHref = drawImages[0].getAttribute("xlink:href") || '';
300
300
  if (imageHref) {
@@ -312,7 +312,7 @@ const parseOpenOffice = async (buffer, config) => {
312
312
  }
313
313
  };
314
314
  if (config.includeRawContent) {
315
- imageNode.rawContent = frame.toString();
315
+ imageNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frame, sourceXml, config);
316
316
  }
317
317
  children.push(imageNode);
318
318
  }
@@ -330,14 +330,14 @@ const parseOpenOffice = async (buffer, config) => {
330
330
  * @param config - Parser configuration
331
331
  * @returns Object containing text, children, alignment, and style info
332
332
  */
333
- const parseParagraphContent = (node, paraStyleMap, styleMap, config) => {
333
+ const parseParagraphContent = (node, paraStyleMap, styleMap, config, sourceXml) => {
334
334
  // Get paragraph style for alignment and drop caps
335
335
  const paraStyle = node.getAttribute("text:style-name");
336
336
  const styleInfo = paraStyle ? paraStyleMap[paraStyle] : undefined;
337
337
  const alignment = styleInfo?.alignment;
338
338
  const dropCap = styleInfo?.dropCap;
339
339
  // Parse content recursively using the new helper
340
- const content = parseInlineContent(node, styleMap, config, notes, paraStyleMap);
340
+ const content = parseInlineContent(node, styleMap, config, notes, paraStyleMap, {}, undefined, sourceXml);
341
341
  // Add style name to metadata of children if they don't have one
342
342
  if (paraStyle) {
343
343
  content.children.forEach(child => {
@@ -401,15 +401,15 @@ const parseOpenOffice = async (buffer, config) => {
401
401
  * @param config - Parser configuration
402
402
  * @returns Table content node with proper structure
403
403
  */
404
- const parseTable = (tableNode, paraStyleMap, styleMap, config) => {
404
+ const parseTable = (tableNode, paraStyleMap, styleMap, config, sourceXml) => {
405
405
  const rows = [];
406
406
  // Use getDirectChildren to avoid nested table rows
407
- const tableRows = (0, xmlUtils_1.getDirectChildren)(tableNode, "table:table-row");
407
+ const tableRows = (0, xmlUtils_js_1.getDirectChildren)(tableNode, "table:table-row");
408
408
  let rowIndex = 0;
409
409
  for (const row of tableRows) {
410
410
  const cells = [];
411
411
  // Use getDirectChildren to avoid nested table cells
412
- const tableCells = (0, xmlUtils_1.getDirectChildren)(row, "table:table-cell");
412
+ const tableCells = (0, xmlUtils_js_1.getDirectChildren)(row, "table:table-cell");
413
413
  const rowsRepeated = parseInt(row.getAttribute("table:number-rows-repeated") || "1");
414
414
  let colIndex = 0;
415
415
  for (const cell of tableCells) {
@@ -424,10 +424,10 @@ const parseOpenOffice = async (buffer, config) => {
424
424
  return;
425
425
  for (let i = 0; i < node.childNodes.length; i++) {
426
426
  const child = node.childNodes[i];
427
- if (child.nodeType === 1) { // Element
427
+ if ((0, xmlUtils_js_1.isElement)(child)) { // Element
428
428
  const element = child;
429
429
  if (element.tagName === "text:p" || element.tagName === "text:h") {
430
- const pContent = parseParagraphContent(element, paraStyleMap, styleMap, config);
430
+ const pContent = parseParagraphContent(element, paraStyleMap, styleMap, config, sourceXml);
431
431
  const pNode = {
432
432
  type: element.tagName === "text:h" ? 'heading' : 'paragraph',
433
433
  text: pContent.text,
@@ -446,7 +446,7 @@ const parseOpenOffice = async (buffer, config) => {
446
446
  pNode.metadata.level = parseInt(element.getAttribute("text:outline-level") || "1");
447
447
  }
448
448
  if (config.includeRawContent) {
449
- pNode.rawContent = element.toString();
449
+ pNode.rawContent = (0, xmlUtils_js_1.getRawContent)(element, sourceXml, config);
450
450
  }
451
451
  cellChildren.push(pNode);
452
452
  cellTextRef.value += pContent.text;
@@ -457,7 +457,7 @@ const parseOpenOffice = async (buffer, config) => {
457
457
  }
458
458
  else if (element.tagName === "table:table") {
459
459
  // Recursive call for nested table
460
- const nestedTableNode = parseTable(element, paraStyleMap, styleMap, config);
460
+ const nestedTableNode = parseTable(element, paraStyleMap, styleMap, config, sourceXml);
461
461
  cellChildren.push(nestedTableNode);
462
462
  }
463
463
  else if (element.tagName === "draw:frame" || element.tagName === "draw:text-box") {
@@ -487,7 +487,7 @@ const parseOpenOffice = async (buffer, config) => {
487
487
  if (rowSpan > 1)
488
488
  cellMetadata.rowSpan = rowSpan;
489
489
  if (config.includeRawContent) {
490
- cellNode.rawContent = cell.toString();
490
+ cellNode.rawContent = (0, xmlUtils_js_1.getRawContent)(cell, sourceXml, config);
491
491
  }
492
492
  cells.push(cellNode);
493
493
  colIndex++;
@@ -508,7 +508,7 @@ const parseOpenOffice = async (buffer, config) => {
508
508
  });
509
509
  }
510
510
  if (config.includeRawContent) {
511
- rowNode.rawContent = row.toString();
511
+ rowNode.rawContent = (0, xmlUtils_js_1.getRawContent)(row, sourceXml, config);
512
512
  }
513
513
  rows.push(rowNode);
514
514
  rowIndex++;
@@ -520,18 +520,20 @@ const parseOpenOffice = async (buffer, config) => {
520
520
  };
521
521
  };
522
522
  const parseContentXml = (xmlString) => {
523
- const xml = (0, xmlUtils_1.parseXmlString)(xmlString);
524
- const body = (0, xmlUtils_1.getElementsByTagName)(xml, "office:body")[0];
523
+ const xml = (0, xmlUtils_js_1.parseXmlString)(xmlString, { locator: config.includeRawContent });
524
+ const body = (0, xmlUtils_js_1.getFirstElementByTagName)(xml, "office:body");
525
+ if (!body)
526
+ return;
525
527
  // Parse automatic styles (local to content.xml)
526
- const automaticStyles = (0, xmlUtils_1.getElementsByTagName)(xml, "office:automatic-styles")[0];
528
+ const automaticStyles = (0, xmlUtils_js_1.getFirstElementByTagName)(xml, "office:automatic-styles");
527
529
  if (automaticStyles) {
528
- const styles = (0, xmlUtils_1.getElementsByTagName)(automaticStyles, "style:style");
530
+ const styles = (0, xmlUtils_js_1.getElementsByTagName)(automaticStyles, "style:style");
529
531
  for (const style of styles) {
530
532
  const name = style.getAttribute("style:name");
531
533
  if (!name)
532
534
  continue;
533
535
  // Parse paragraph properties for alignment
534
- const paraProps = (0, xmlUtils_1.getElementsByTagName)(style, "style:paragraph-properties")[0];
536
+ const paraProps = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "style:paragraph-properties");
535
537
  const styleInfo = {};
536
538
  if (paraProps) {
537
539
  const textAlign = paraProps.getAttribute("fo:text-align");
@@ -548,14 +550,14 @@ const parseOpenOffice = async (buffer, config) => {
548
550
  styleInfo.alignment = alignMap[textAlign];
549
551
  }
550
552
  }
551
- const dropCap = (0, xmlUtils_1.getElementsByTagName)(paraProps, "style:drop-cap")[0];
553
+ const dropCap = (0, xmlUtils_js_1.getFirstElementByTagName)(paraProps, "style:drop-cap");
552
554
  if (dropCap)
553
555
  styleInfo.dropCap = true;
554
556
  }
555
557
  if (Object.keys(styleInfo).length > 0) {
556
558
  paragraphStyleMap[name] = styleInfo;
557
559
  }
558
- const textProps = (0, xmlUtils_1.getElementsByTagName)(style, "style:text-properties")[0];
560
+ const textProps = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "style:text-properties");
559
561
  if (textProps) {
560
562
  const formatting = {};
561
563
  if (textProps.getAttribute("fo:font-weight") === "bold" || textProps.getAttribute("style:font-weight-asian") === "bold")
@@ -593,6 +595,19 @@ const parseOpenOffice = async (buffer, config) => {
593
595
  }
594
596
  }
595
597
  }
598
+ // Start traversal
599
+ const officeBody = (0, xmlUtils_js_1.getFirstElementByTagName)(xml, "office:body");
600
+ if (officeBody) {
601
+ const bodyContent = (0, xmlUtils_js_1.getDirectChildren)(officeBody, "office:text")[0] ||
602
+ (0, xmlUtils_js_1.getDirectChildren)(officeBody, "office:presentation")[0] ||
603
+ (0, xmlUtils_js_1.getDirectChildren)(officeBody, "office:spreadsheet")[0];
604
+ if (bodyContent) {
605
+ const bodyChildren = (0, xmlUtils_js_1.getDirectChildren)(bodyContent, "*");
606
+ for (const child of bodyChildren) {
607
+ traverse(child, content, false, xmlString);
608
+ }
609
+ }
610
+ }
596
611
  /**
597
612
  * Recursively traverses a node and its children to extract content.
598
613
  * Properly handles paragraphs, headings, tables, lists, and frames.
@@ -600,10 +615,11 @@ const parseOpenOffice = async (buffer, config) => {
600
615
  * @param node - The element to traverse
601
616
  * @param targetArray - The array to push extracted content nodes to
602
617
  * @param forceHeading - If true, treats all paragraphs as headings (used for slide titles)
618
+ * @param sourceXml - The source XML string for raw content extraction
603
619
  */
604
- const traverse = (node, targetArray, forceHeading = false) => {
620
+ function traverse(node, targetArray, forceHeading = false, sourceXml) {
605
621
  if (node.tagName === "text:p") {
606
- const pContent = parseParagraphContent(node, paragraphStyleMap, styleMap, config);
622
+ const pContent = parseParagraphContent(node, paragraphStyleMap, styleMap, config, sourceXml);
607
623
  const type = (forceHeading || (node.getAttribute("text:style-name") || '').toLowerCase().includes('title')) ? 'heading' : 'paragraph';
608
624
  const pNode = {
609
625
  type,
@@ -621,13 +637,13 @@ const parseOpenOffice = async (buffer, config) => {
621
637
  if (Object.keys(pNode.metadata || {}).length === 0)
622
638
  delete pNode.metadata;
623
639
  if (config.includeRawContent) {
624
- pNode.rawContent = node.toString();
640
+ pNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
625
641
  }
626
642
  targetArray.push(pNode);
627
643
  }
628
644
  else if (node.tagName === "text:h") {
629
645
  const level = parseInt(node.getAttribute("text:outline-level") || "1");
630
- const hContent = parseParagraphContent(node, paragraphStyleMap, styleMap, config);
646
+ const hContent = parseParagraphContent(node, paragraphStyleMap, styleMap, config, sourceXml);
631
647
  const hNode = {
632
648
  type: 'heading',
633
649
  text: hContent.text,
@@ -639,21 +655,21 @@ const parseOpenOffice = async (buffer, config) => {
639
655
  }
640
656
  };
641
657
  if (config.includeRawContent) {
642
- hNode.rawContent = node.toString();
658
+ hNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
643
659
  }
644
660
  targetArray.push(hNode);
645
661
  }
646
662
  else if (node.tagName === "table:table") {
647
663
  // Parse table with proper structure
648
- const tableNode = parseTable(node, paragraphStyleMap, styleMap, config);
664
+ const tableNode = parseTable(node, paragraphStyleMap, styleMap, config, sourceXml);
649
665
  if (config.includeRawContent) {
650
- tableNode.rawContent = node.toString();
666
+ tableNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
651
667
  }
652
668
  targetArray.push(tableNode);
653
669
  }
654
670
  else if (node.tagName === "text:list") {
655
671
  // Parse list structure with proper listId tracking
656
- const listItems = (0, xmlUtils_1.getDirectChildren)(node, "text:list-item");
672
+ const listItems = (0, xmlUtils_js_1.getDirectChildren)(node, "text:list-item");
657
673
  // Get list style name to use as listId (or generate one)
658
674
  const listStyleName = node.getAttribute("text:style-name") || node.getAttribute("xml:id");
659
675
  const listId = listStyleName || `list-${targetArray.length}`;
@@ -675,15 +691,15 @@ const parseOpenOffice = async (buffer, config) => {
675
691
  }
676
692
  // Try to find list style in automatic styles to determine type and visibility
677
693
  if (styleNameToCheck) {
678
- const automaticStyles = (0, xmlUtils_1.getElementsByTagName)((0, xmlUtils_1.parseXmlString)(mainContentFile?.content.toString() || ''), "office:automatic-styles")[0];
694
+ const automaticStyles = (0, xmlUtils_js_1.getFirstElementByTagName)((0, xmlUtils_js_1.parseXmlString)(mainContentFile?.content.toString() || ''), "office:automatic-styles");
679
695
  if (automaticStyles) {
680
- const listStyles = (0, xmlUtils_1.getElementsByTagName)(automaticStyles, "text:list-style");
696
+ const listStyles = (0, xmlUtils_js_1.getElementsByTagName)(automaticStyles, "text:list-style");
681
697
  for (const listStyle of listStyles) {
682
698
  if (listStyle.getAttribute("style:name") === styleNameToCheck) {
683
699
  // Check if it has bullet or number level styles
684
- const bulletLevels = (0, xmlUtils_1.getElementsByTagName)(listStyle, "text:list-level-style-bullet");
685
- const numberLevels = (0, xmlUtils_1.getElementsByTagName)(listStyle, "text:list-level-style-number");
686
- const imageLevels = (0, xmlUtils_1.getElementsByTagName)(listStyle, "text:list-level-style-image");
700
+ const bulletLevels = (0, xmlUtils_js_1.getElementsByTagName)(listStyle, "text:list-level-style-bullet");
701
+ const numberLevels = (0, xmlUtils_js_1.getElementsByTagName)(listStyle, "text:list-level-style-number");
702
+ const imageLevels = (0, xmlUtils_js_1.getElementsByTagName)(listStyle, "text:list-level-style-image");
687
703
  if (numberLevels.length > 0) {
688
704
  listType = 'ordered';
689
705
  isVisible = numberLevels.some(l => !!l.getAttribute("style:num-format"));
@@ -700,13 +716,13 @@ const parseOpenOffice = async (buffer, config) => {
700
716
  }
701
717
  // Also check in styles.xml if still unordered and hidden
702
718
  if (stylesFile && !isVisible) {
703
- const stylesXml = (0, xmlUtils_1.parseXmlString)(stylesFile.content.toString());
704
- const listStyles = (0, xmlUtils_1.getElementsByTagName)(stylesXml, "text:list-style");
719
+ const stylesXml = (0, xmlUtils_js_1.parseXmlString)(stylesFile.content.toString());
720
+ const listStyles = (0, xmlUtils_js_1.getElementsByTagName)(stylesXml, "text:list-style");
705
721
  for (const listStyle of listStyles) {
706
722
  if (listStyle.getAttribute("style:name") === styleNameToCheck) {
707
- const bulletLevels = (0, xmlUtils_1.getElementsByTagName)(listStyle, "text:list-level-style-bullet");
708
- const numberLevels = (0, xmlUtils_1.getElementsByTagName)(listStyle, "text:list-level-style-number");
709
- const imageLevels = (0, xmlUtils_1.getElementsByTagName)(listStyle, "text:list-level-style-image");
723
+ const bulletLevels = (0, xmlUtils_js_1.getElementsByTagName)(listStyle, "text:list-level-style-bullet");
724
+ const numberLevels = (0, xmlUtils_js_1.getElementsByTagName)(listStyle, "text:list-level-style-number");
725
+ const imageLevels = (0, xmlUtils_js_1.getElementsByTagName)(listStyle, "text:list-level-style-image");
710
726
  if (numberLevels.length > 0) {
711
727
  listType = 'ordered';
712
728
  isVisible = numberLevels.some(l => !!l.getAttribute("style:num-format"));
@@ -730,8 +746,8 @@ const parseOpenOffice = async (buffer, config) => {
730
746
  if (item.childNodes) {
731
747
  for (let j = 0; j < item.childNodes.length; j++) {
732
748
  const child = item.childNodes[j];
733
- if (child.nodeType === 1) { // Element
734
- traverse(child, targetArray, forceHeading);
749
+ if ((0, xmlUtils_js_1.isElement)(child)) { // Element
750
+ traverse(child, targetArray, forceHeading, sourceXml);
735
751
  }
736
752
  }
737
753
  }
@@ -771,10 +787,10 @@ const parseOpenOffice = async (buffer, config) => {
771
787
  if (item.childNodes) {
772
788
  for (let j = 0; j < item.childNodes.length; j++) {
773
789
  const child = item.childNodes[j];
774
- if (child.nodeType === 1) { // Element
790
+ if ((0, xmlUtils_js_1.isElement)(child)) { // Element
775
791
  const element = child;
776
792
  if (element.tagName === "text:p") {
777
- const pContent = parseParagraphContent(element, paragraphStyleMap, styleMap, config);
793
+ const pContent = parseParagraphContent(element, paragraphStyleMap, styleMap, config, sourceXml);
778
794
  const listNode = {
779
795
  type: 'list',
780
796
  text: pContent.text,
@@ -789,12 +805,12 @@ const parseOpenOffice = async (buffer, config) => {
789
805
  }
790
806
  };
791
807
  if (config.includeRawContent)
792
- listNode.rawContent = element.toString();
808
+ listNode.rawContent = (0, xmlUtils_js_1.getRawContent)(element, sourceXml, config);
793
809
  targetArray.push(listNode);
794
810
  }
795
811
  else if (element.tagName === "text:h") {
796
812
  const level = parseInt(element.getAttribute("text:outline-level") || "1");
797
- const hContent = parseParagraphContent(element, paragraphStyleMap, styleMap, config);
813
+ const hContent = parseParagraphContent(element, paragraphStyleMap, styleMap, config, sourceXml);
798
814
  const listNode = {
799
815
  type: 'list',
800
816
  text: hContent.text,
@@ -809,12 +825,12 @@ const parseOpenOffice = async (buffer, config) => {
809
825
  }
810
826
  };
811
827
  if (config.includeRawContent)
812
- listNode.rawContent = element.toString();
828
+ listNode.rawContent = (0, xmlUtils_js_1.getRawContent)(element, sourceXml, config);
813
829
  targetArray.push(listNode);
814
830
  }
815
831
  else if (element.tagName === "text:list") {
816
832
  // Recursive call for nested list
817
- traverse(element, targetArray, forceHeading);
833
+ traverse(element, targetArray, forceHeading, sourceXml);
818
834
  }
819
835
  }
820
836
  }
@@ -825,24 +841,24 @@ const parseOpenOffice = async (buffer, config) => {
825
841
  const presClass = node.getAttribute("presentation:class");
826
842
  const isHeading = presClass === "title" || presClass === "sub-title";
827
843
  // In presentations, frames often contain text-boxes, images, tables, or objects
828
- const textBox = (0, xmlUtils_1.getElementsByTagName)(node, "draw:text-box")[0];
829
- const image = (0, xmlUtils_1.getElementsByTagName)(node, "draw:image")[0];
830
- const table = (0, xmlUtils_1.getElementsByTagName)(node, "table:table")[0];
831
- const object = (0, xmlUtils_1.getElementsByTagName)(node, "draw:object")[0];
844
+ const textBox = (0, xmlUtils_js_1.getFirstElementByTagName)(node, "draw:text-box");
845
+ const image = (0, xmlUtils_js_1.getFirstElementByTagName)(node, "draw:image");
846
+ const table = (0, xmlUtils_js_1.getFirstElementByTagName)(node, "table:table");
847
+ const object = (0, xmlUtils_js_1.getFirstElementByTagName)(node, "draw:object");
832
848
  if (textBox) {
833
- traverse(textBox, targetArray, isHeading || forceHeading);
849
+ traverse(textBox, targetArray, isHeading || forceHeading, sourceXml);
834
850
  }
835
851
  else if (table) {
836
- const tableNode = parseTable(table, paragraphStyleMap, styleMap, config);
852
+ const tableNode = parseTable(table, paragraphStyleMap, styleMap, config, sourceXml);
837
853
  if (config.includeRawContent)
838
- tableNode.rawContent = table.toString();
854
+ tableNode.rawContent = (0, xmlUtils_js_1.getRawContent)(table, sourceXml, config);
839
855
  targetArray.push(tableNode);
840
856
  }
841
857
  else if (image) {
842
858
  // Extract alt text from svg:title or svg:desc
843
859
  let altText = '';
844
- const svgTitle = (0, xmlUtils_1.getElementsByTagName)(node, "svg:title")[0];
845
- const svgDesc = (0, xmlUtils_1.getElementsByTagName)(node, "svg:desc")[0];
860
+ const svgTitle = (0, xmlUtils_js_1.getFirstElementByTagName)(node, "svg:title");
861
+ const svgDesc = (0, xmlUtils_js_1.getFirstElementByTagName)(node, "svg:desc");
846
862
  if (svgTitle && svgTitle.textContent) {
847
863
  altText = svgTitle.textContent;
848
864
  }
@@ -865,7 +881,7 @@ const parseOpenOffice = async (buffer, config) => {
865
881
  }
866
882
  };
867
883
  if (config.includeRawContent) {
868
- imageNode.rawContent = node.toString();
884
+ imageNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
869
885
  }
870
886
  targetArray.push(imageNode);
871
887
  }
@@ -877,7 +893,7 @@ const parseOpenOffice = async (buffer, config) => {
877
893
  const objectPath = `${attachmentName}/content.xml`;
878
894
  const objectFile = files.find(f => f.path === objectPath || f.path.endsWith(objectPath));
879
895
  if (objectFile) {
880
- const chartData = (0, chartUtils_1.extractChartData)(objectFile.content);
896
+ const chartData = (0, chartUtils_js_1.extractChartData)(objectFile.content);
881
897
  const chartNode = {
882
898
  type: 'chart',
883
899
  text: chartData.rawTexts.join(" "),
@@ -887,7 +903,7 @@ const parseOpenOffice = async (buffer, config) => {
887
903
  }
888
904
  };
889
905
  if (config.includeRawContent)
890
- chartNode.rawContent = node.toString();
906
+ chartNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
891
907
  targetArray.push(chartNode);
892
908
  }
893
909
  else {
@@ -897,7 +913,7 @@ const parseOpenOffice = async (buffer, config) => {
897
913
  metadata: { attachmentName: attachmentName }
898
914
  };
899
915
  if (config.includeRawContent)
900
- chartNode.rawContent = node.toString();
916
+ chartNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
901
917
  targetArray.push(chartNode);
902
918
  }
903
919
  }
@@ -907,28 +923,29 @@ const parseOpenOffice = async (buffer, config) => {
907
923
  if (node.childNodes) {
908
924
  for (let i = 0; i < node.childNodes.length; i++) {
909
925
  const child = node.childNodes[i];
910
- if (child.nodeType === 1) { // Element
911
- traverse(child, targetArray, forceHeading);
926
+ if ((0, xmlUtils_js_1.isElement)(child)) { // Element
927
+ traverse(child, targetArray, forceHeading, sourceXml);
912
928
  }
913
929
  }
914
930
  }
915
931
  }
916
- };
932
+ }
933
+ ;
917
934
  // ODS: Spreadsheet
918
935
  if (fileType === 'ods') {
919
- const spreadsheet = (0, xmlUtils_1.getElementsByTagName)(body, "office:spreadsheet")[0];
936
+ const spreadsheet = (0, xmlUtils_js_1.getFirstElementByTagName)(body, "office:spreadsheet");
920
937
  if (spreadsheet) {
921
- const tables = (0, xmlUtils_1.getElementsByTagName)(spreadsheet, "table:table");
938
+ const tables = (0, xmlUtils_js_1.getElementsByTagName)(spreadsheet, "table:table");
922
939
  for (let i = 0; i < tables.length; i++) {
923
940
  const table = tables[i];
924
941
  const sheetName = table.getAttribute("table:name") || `Sheet${i + 1}`;
925
942
  const rows = [];
926
- const tableRows = (0, xmlUtils_1.getElementsByTagName)(table, "table:table-row");
943
+ const tableRows = (0, xmlUtils_js_1.getElementsByTagName)(table, "table:table-row");
927
944
  let rowIndex = 0;
928
945
  for (let r = 0; r < tableRows.length; r++) {
929
946
  const row = tableRows[r];
930
947
  const cells = [];
931
- const tableCells = (0, xmlUtils_1.getElementsByTagName)(row, "table:table-cell");
948
+ const tableCells = (0, xmlUtils_js_1.getElementsByTagName)(row, "table:table-cell");
932
949
  let colIndex = 0;
933
950
  const rowsRepeated = parseInt(row.getAttribute("table:number-rows-repeated") || "1");
934
951
  for (let c = 0; c < tableCells.length; c++) {
@@ -937,11 +954,11 @@ const parseOpenOffice = async (buffer, config) => {
937
954
  // Extract text from cell (paragraphs inside cell)
938
955
  let cellText = "";
939
956
  const children = [];
940
- const ps = (0, xmlUtils_1.getElementsByTagName)(cell, "text:p");
957
+ const ps = (0, xmlUtils_js_1.getElementsByTagName)(cell, "text:p");
941
958
  for (let p = 0; p < ps.length; p++) {
942
959
  const para = ps[p];
943
960
  // Parse text:span elements for formatted text
944
- const spans = (0, xmlUtils_1.getElementsByTagName)(para, "text:span");
961
+ const spans = (0, xmlUtils_js_1.getElementsByTagName)(para, "text:span");
945
962
  if (spans.length > 0) {
946
963
  for (const span of spans) {
947
964
  const styleName = span.getAttribute("text:style-name");
@@ -973,12 +990,12 @@ const parseOpenOffice = async (buffer, config) => {
973
990
  cellText += "\n";
974
991
  }
975
992
  // Check for embedded draw:frame (images) in cell
976
- const drawFrames = (0, xmlUtils_1.getElementsByTagName)(cell, "draw:frame");
993
+ const drawFrames = (0, xmlUtils_js_1.getElementsByTagName)(cell, "draw:frame");
977
994
  for (const frame of drawFrames) {
978
995
  // Extract alt text from svg:title or svg:desc
979
996
  let altText = '';
980
- const svgTitle = (0, xmlUtils_1.getElementsByTagName)(frame, "svg:title")[0];
981
- const svgDesc = (0, xmlUtils_1.getElementsByTagName)(frame, "svg:desc")[0];
997
+ const svgTitle = (0, xmlUtils_js_1.getFirstElementByTagName)(frame, "svg:title");
998
+ const svgDesc = (0, xmlUtils_js_1.getFirstElementByTagName)(frame, "svg:desc");
982
999
  if (svgTitle && svgTitle.textContent) {
983
1000
  altText = svgTitle.textContent;
984
1001
  }
@@ -987,7 +1004,7 @@ const parseOpenOffice = async (buffer, config) => {
987
1004
  }
988
1005
  // Extract image href
989
1006
  let imageHref = '';
990
- const drawImages = (0, xmlUtils_1.getElementsByTagName)(frame, "draw:image");
1007
+ const drawImages = (0, xmlUtils_js_1.getElementsByTagName)(frame, "draw:image");
991
1008
  if (drawImages.length > 0) {
992
1009
  const rawHref = drawImages[0].getAttribute("xlink:href");
993
1010
  if (rawHref) {
@@ -997,7 +1014,7 @@ const parseOpenOffice = async (buffer, config) => {
997
1014
  }
998
1015
  // Extract chart object href
999
1016
  let chartHref = '';
1000
- const drawObjects = (0, xmlUtils_1.getElementsByTagName)(frame, "draw:object");
1017
+ const drawObjects = (0, xmlUtils_js_1.getElementsByTagName)(frame, "draw:object");
1001
1018
  if (drawObjects.length > 0) {
1002
1019
  const href = drawObjects[0].getAttribute("xlink:href");
1003
1020
  if (href) {
@@ -1017,7 +1034,7 @@ const parseOpenOffice = async (buffer, config) => {
1017
1034
  }
1018
1035
  };
1019
1036
  if (config.includeRawContent) {
1020
- imageNode.rawContent = frame.toString();
1037
+ imageNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frame, xmlString, config);
1021
1038
  }
1022
1039
  children.push(imageNode);
1023
1040
  }
@@ -1046,7 +1063,7 @@ const parseOpenOffice = async (buffer, config) => {
1046
1063
  metadata: { row: rowIndex, col: colIndex }
1047
1064
  };
1048
1065
  if (config.includeRawContent) {
1049
- cellNode.rawContent = cell.toString();
1066
+ cellNode.rawContent = (0, xmlUtils_js_1.getRawContent)(cell, xmlString, config);
1050
1067
  }
1051
1068
  cells.push(cellNode);
1052
1069
  }
@@ -1070,7 +1087,7 @@ const parseOpenOffice = async (buffer, config) => {
1070
1087
  });
1071
1088
  }
1072
1089
  if (config.includeRawContent) {
1073
- rowNode.rawContent = row.toString();
1090
+ rowNode.rawContent = (0, xmlUtils_js_1.getRawContent)(row, xmlString, config);
1074
1091
  }
1075
1092
  rows.push(rowNode);
1076
1093
  rowIndex++;
@@ -1086,7 +1103,7 @@ const parseOpenOffice = async (buffer, config) => {
1086
1103
  metadata: { sheetName }
1087
1104
  };
1088
1105
  if (config.includeRawContent) {
1089
- sheetNode.rawContent = table.toString();
1106
+ sheetNode.rawContent = (0, xmlUtils_js_1.getRawContent)(table, xmlString, config);
1090
1107
  }
1091
1108
  content.push(sheetNode);
1092
1109
  }
@@ -1094,9 +1111,9 @@ const parseOpenOffice = async (buffer, config) => {
1094
1111
  }
1095
1112
  // ODP: Presentation
1096
1113
  else if (fileType === 'odp') {
1097
- const presentation = (0, xmlUtils_1.getElementsByTagName)(body, "office:presentation")[0];
1114
+ const presentation = (0, xmlUtils_js_1.getFirstElementByTagName)(body, "office:presentation");
1098
1115
  if (presentation) {
1099
- const pages = (0, xmlUtils_1.getDirectChildren)(presentation, "draw:page");
1116
+ const pages = (0, xmlUtils_js_1.getDirectChildren)(presentation, "draw:page");
1100
1117
  const odpNotes = [];
1101
1118
  for (let i = 0; i < pages.length; i++) {
1102
1119
  const page = pages[i];
@@ -1111,7 +1128,7 @@ const parseOpenOffice = async (buffer, config) => {
1111
1128
  if (pageChildren) {
1112
1129
  for (let j = 0; j < pageChildren.length; j++) {
1113
1130
  const child = pageChildren[j];
1114
- if (child.nodeType === 1) { // Element
1131
+ if ((0, xmlUtils_js_1.isElement)(child)) { // Element
1115
1132
  const element = child;
1116
1133
  if (element.tagName === "presentation:notes") {
1117
1134
  if (!config.ignoreNotes) {
@@ -1123,16 +1140,16 @@ const parseOpenOffice = async (buffer, config) => {
1123
1140
  noteId: `slide-note-${i + 1}`
1124
1141
  }
1125
1142
  };
1126
- traverse(element, noteNode.children);
1143
+ traverse(element, noteNode.children, false, xmlString);
1127
1144
  }
1128
1145
  continue;
1129
1146
  }
1130
- traverse(element, slideNode.children);
1147
+ traverse(element, slideNode.children, false, xmlString);
1131
1148
  }
1132
1149
  }
1133
1150
  }
1134
1151
  if (config.includeRawContent) {
1135
- slideNode.rawContent = page.toString();
1152
+ slideNode.rawContent = (0, xmlUtils_js_1.getRawContent)(page, xmlString, config);
1136
1153
  }
1137
1154
  content.push(slideNode);
1138
1155
  if (noteNode && noteNode.children && noteNode.children.length > 0) {
@@ -1151,9 +1168,9 @@ const parseOpenOffice = async (buffer, config) => {
1151
1168
  }
1152
1169
  // ODT: Text Document (and generic fallback)
1153
1170
  else {
1154
- const textDoc = (0, xmlUtils_1.getElementsByTagName)(body, "office:text")[0];
1171
+ const textDoc = (0, xmlUtils_js_1.getFirstElementByTagName)(body, "office:text");
1155
1172
  if (textDoc) {
1156
- traverse(textDoc, content);
1173
+ traverse(textDoc, content, false, xmlString);
1157
1174
  }
1158
1175
  }
1159
1176
  };
@@ -1167,8 +1184,8 @@ const parseOpenOffice = async (buffer, config) => {
1167
1184
  if (config.extractAttachments) {
1168
1185
  const objectFiles = files.filter(f => f.path.match(/Object \d+\/content\.xml/));
1169
1186
  for (const objFile of objectFiles) {
1170
- const objXml = (0, xmlUtils_1.parseXmlString)(objFile.content.toString());
1171
- const isChart = (0, xmlUtils_1.getElementsByTagName)(objXml, "chart:chart").length > 0;
1187
+ const objXml = (0, xmlUtils_js_1.parseXmlString)(objFile.content.toString());
1188
+ const isChart = (0, xmlUtils_js_1.getElementsByTagName)(objXml, "chart:chart").length > 0;
1172
1189
  if (isChart) {
1173
1190
  const objectId = objFile.path.split('/')[0];
1174
1191
  const attachment = {
@@ -1179,7 +1196,7 @@ const parseOpenOffice = async (buffer, config) => {
1179
1196
  extension: 'xml'
1180
1197
  };
1181
1198
  // Extract data from chart XML
1182
- const chartData = (0, chartUtils_1.extractChartData)(objFile.content);
1199
+ const chartData = (0, chartUtils_js_1.extractChartData)(objFile.content);
1183
1200
  if (chartData.rawTexts.length > 0) {
1184
1201
  attachment.chartData = chartData;
1185
1202
  }
@@ -1189,22 +1206,22 @@ const parseOpenOffice = async (buffer, config) => {
1189
1206
  }
1190
1207
  if (config.extractAttachments) {
1191
1208
  for (const media of mediaFiles) {
1192
- const attachment = (0, imageUtils_1.createAttachment)(media.path.split('/').pop() || 'image', media.content);
1209
+ const attachment = (0, imageUtils_js_1.createAttachment)(media.path.split('/').pop() || 'image', media.content);
1193
1210
  attachments.push(attachment);
1194
1211
  if (config.ocr) {
1195
1212
  if (attachment.mimeType.startsWith('image/')) {
1196
1213
  try {
1197
- attachment.ocrText = (await (0, ocrUtils_1.performOcr)(media.content, config.ocrLanguage)).trim();
1214
+ attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(media.content, { language: config.ocrLanguage, ...config.ocrConfig })).trim();
1198
1215
  }
1199
1216
  catch (e) {
1200
- (0, errorUtils_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
1217
+ (0, errorUtils_js_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
1201
1218
  }
1202
1219
  }
1203
1220
  }
1204
1221
  }
1205
1222
  }
1206
1223
  const metaFile = files.find(f => f.path.match(metaFileRegex));
1207
- const metadata = metaFile ? (0, xmlUtils_1.parseOfficeMetadata)(metaFile.content.toString()) : {};
1224
+ const metadata = metaFile ? (0, xmlUtils_js_1.parseOfficeMetadata)(metaFile.content.toString()) : {};
1208
1225
  // Helper: Resolve ODS chart cell references to actual values
1209
1226
  // ODS charts often link to cell ranges (e.g., [Sheet1.$A$1:.$A$5]) instead of embedding values
1210
1227
  const resolveChartReferences = (chartData, nodes) => {