officeparser 6.1.0 → 6.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -422,10 +422,10 @@ const parseExcel = async (buffer, config) => {
422
422
  if (cMatches) {
423
423
  for (const cXml of cMatches) {
424
424
  // Extract cell value
425
- const typeMatch = cXml.match(/t="([a-z]+)"/);
425
+ const typeMatch = cXml.match(/t="([a-zA-Z]+)"/);
426
426
  const type = typeMatch ? typeMatch[1] : 'n'; // n = number (default)
427
- const vMatch = cXml.match(/<v>(.*?)<\/v>/);
428
- const tMatch = cXml.match(/<t>(.*?)<\/t>/);
427
+ const vMatch = cXml.match(/<v>([\s\S]*?)<\/v>/);
428
+ const tMatch = cXml.match(/<t>([\s\S]*?)<\/t>/);
429
429
  let text = '';
430
430
  let cellNodes = [];
431
431
  if (type === 's' && vMatch) {
@@ -442,10 +442,10 @@ const parseExcel = async (buffer, config) => {
442
442
  }
443
443
  }
444
444
  else if (type === 'inlineStr' && tMatch) {
445
- text = tMatch[1];
445
+ text = tMatch[1].trim();
446
446
  }
447
447
  else if (vMatch) {
448
- text = vMatch[1];
448
+ text = vMatch[1].trim();
449
449
  }
450
450
  // Parse cell coordinate
451
451
  const coordMatch = cXml.match(/r="([A-Z]+)(\d+)"/);
@@ -70,6 +70,11 @@ const parseOpenOffice = async (buffer, config) => {
70
70
  const styleMap = {};
71
71
  const paragraphStyleMap = {};
72
72
  const listCounters = {}; // Track item index per listId/level
73
+ let currentListId = null;
74
+ let lastListType = null;
75
+ let lastListStyle = null;
76
+ let listIdCounter = 0;
77
+ let lastWasList = false;
73
78
  // Helper to parse styles
74
79
  const parseStyles = (xmlString) => {
75
80
  const xml = (0, xmlUtils_js_1.parseXmlString)(xmlString);
@@ -221,7 +226,7 @@ const parseOpenOffice = async (buffer, config) => {
221
226
  type: 'text',
222
227
  text: '\n',
223
228
  formatting: parentFormatting,
224
- metadata: linkMetadata ? { ...linkMetadata } : undefined
229
+ metadata: { ...(linkMetadata || {}), isLineBreak: true }
225
230
  });
226
231
  }
227
232
  else if (tagName === 'text:span') {
@@ -391,6 +396,31 @@ const parseOpenOffice = async (buffer, config) => {
391
396
  }
392
397
  return { text: content.text, children: content.children, alignment, style: paraStyle || undefined };
393
398
  };
399
+ /**
400
+ * Splits paragraph content into multiple segments based on line breaks.
401
+ * Used to handle soft line breaks within list items.
402
+ *
403
+ * @param pContent - The content of a single paragraph
404
+ * @returns Array of content segments
405
+ */
406
+ const splitParagraphByBreaks = (pContent) => {
407
+ const segments = [];
408
+ let currentText = "";
409
+ let currentChildren = [];
410
+ for (const child of pContent.children) {
411
+ if (child.type === "text" && child.metadata?.isLineBreak) {
412
+ segments.push({ text: currentText, children: currentChildren });
413
+ currentText = "";
414
+ currentChildren = [];
415
+ }
416
+ else {
417
+ currentText += child.text || "";
418
+ currentChildren.push(child);
419
+ }
420
+ }
421
+ segments.push({ text: currentText, children: currentChildren });
422
+ return segments;
423
+ };
394
424
  /**
395
425
  * Helper to parse a table node and extract its structure.
396
426
  * Properly creates table → row → cell hierarchy with metadata.
@@ -640,6 +670,7 @@ const parseOpenOffice = async (buffer, config) => {
640
670
  pNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
641
671
  }
642
672
  targetArray.push(pNode);
673
+ lastWasList = false;
643
674
  }
644
675
  else if (node.tagName === "text:h") {
645
676
  const level = parseInt(node.getAttribute("text:outline-level") || "1");
@@ -658,6 +689,7 @@ const parseOpenOffice = async (buffer, config) => {
658
689
  hNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
659
690
  }
660
691
  targetArray.push(hNode);
692
+ lastWasList = false;
661
693
  }
662
694
  else if (node.tagName === "table:table") {
663
695
  // Parse table with proper structure
@@ -666,16 +698,15 @@ const parseOpenOffice = async (buffer, config) => {
666
698
  tableNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
667
699
  }
668
700
  targetArray.push(tableNode);
701
+ lastWasList = false;
669
702
  }
670
703
  else if (node.tagName === "text:list") {
671
704
  // Parse list structure with proper listId tracking
672
705
  const listItems = (0, xmlUtils_js_1.getDirectChildren)(node, "text:list-item");
673
- // Get list style name to use as listId (or generate one)
674
- const listStyleName = node.getAttribute("text:style-name") || node.getAttribute("xml:id");
675
- const listId = listStyleName || `list-${targetArray.length}`;
676
706
  // Determine list type by checking the list style definition
677
707
  let listType = 'unordered';
678
708
  let isVisible = false;
709
+ const listStyleName = node.getAttribute("text:style-name") || node.getAttribute("xml:id");
679
710
  let styleNameToCheck = listStyleName;
680
711
  // If no style name, check parent list for inherited style
681
712
  if (!styleNameToCheck) {
@@ -741,6 +772,7 @@ const parseOpenOffice = async (buffer, config) => {
741
772
  // If the list is not visible, it's likely a layout list used by Impress.
742
773
  // We should traverse its items and treat their content as regular nodes.
743
774
  if (!isVisible) {
775
+ lastWasList = false;
744
776
  for (let i = 0; i < listItems.length; i++) {
745
777
  const item = listItems[i];
746
778
  if (item.childNodes) {
@@ -754,6 +786,24 @@ const parseOpenOffice = async (buffer, config) => {
754
786
  }
755
787
  return;
756
788
  }
789
+ // List Continuity Logic:
790
+ // If this list follows another list of the same type and style, or we are in ODP and it's sequential,
791
+ // we should reuse the previous listId to maintain numbering.
792
+ const isODP = fileType === 'odp';
793
+ const sameStyle = styleNameToCheck && styleNameToCheck === lastListStyle;
794
+ const sameType = listType === lastListType;
795
+ let listId;
796
+ if (lastWasList && (sameStyle || (isODP && sameType))) {
797
+ listId = currentListId;
798
+ }
799
+ else {
800
+ // New list
801
+ listId = styleNameToCheck || `list-${++listIdCounter}`;
802
+ currentListId = listId;
803
+ lastListType = listType;
804
+ lastListStyle = styleNameToCheck;
805
+ }
806
+ lastWasList = true;
757
807
  // Calculate indentation level by counting parent text:list elements
758
808
  let indentation = 0;
759
809
  let parent = node.parentNode;
@@ -774,59 +824,57 @@ const parseOpenOffice = async (buffer, config) => {
774
824
  // Process each list item
775
825
  for (let i = 0; i < listItems.length; i++) {
776
826
  const item = listItems[i];
777
- // Increment item index for this list/level
778
- listCounters[listId][indentKey]++;
779
- const itemIndex = listCounters[listId][indentKey];
780
- // Reset deeper levels when we encounter an item at this level
781
- for (let k = indentation + 1; k < 10; k++) {
782
- if (listCounters[listId][k.toString()] !== undefined) {
783
- listCounters[listId][k.toString()] = -1;
784
- }
785
- }
827
+ let hasIndexedThisItem = false;
786
828
  // Iterate over direct children of list item (paragraphs, headings, nested lists)
787
829
  if (item.childNodes) {
788
830
  for (let j = 0; j < item.childNodes.length; j++) {
789
831
  const child = item.childNodes[j];
790
832
  if ((0, xmlUtils_js_1.isElement)(child)) { // Element
791
833
  const element = child;
792
- if (element.tagName === "text:p") {
793
- const pContent = parseParagraphContent(element, paragraphStyleMap, styleMap, config, sourceXml);
794
- const listNode = {
795
- type: 'list',
796
- text: pContent.text,
797
- children: pContent.children,
798
- metadata: {
799
- listType,
800
- indentation,
801
- itemIndex,
802
- listId,
803
- alignment: pContent.alignment || 'left',
804
- style: pContent.style
834
+ if (element.tagName === "text:p" || element.tagName === "text:h") {
835
+ if (!hasIndexedThisItem) {
836
+ listCounters[listId][indentKey]++;
837
+ hasIndexedThisItem = true;
838
+ for (let k = indentation + 1; k < 10; k++) {
839
+ if (listCounters[listId][k.toString()] !== undefined) {
840
+ listCounters[listId][k.toString()] = -1;
841
+ }
805
842
  }
806
- };
807
- if (config.includeRawContent)
808
- listNode.rawContent = (0, xmlUtils_js_1.getRawContent)(element, sourceXml, config);
809
- targetArray.push(listNode);
810
- }
811
- else if (element.tagName === "text:h") {
812
- const level = parseInt(element.getAttribute("text:outline-level") || "1");
813
- const hContent = parseParagraphContent(element, paragraphStyleMap, styleMap, config, sourceXml);
814
- const listNode = {
815
- type: 'list',
816
- text: hContent.text,
817
- children: hContent.children,
818
- metadata: {
819
- listType,
820
- indentation,
821
- itemIndex,
822
- listId,
823
- ...(hContent.alignment ? { alignment: hContent.alignment } : {}),
824
- style: hContent.style
843
+ }
844
+ const itemIndex = listCounters[listId][indentKey];
845
+ const pContent = parseParagraphContent(element, paragraphStyleMap, styleMap, config, sourceXml);
846
+ const segments = splitParagraphByBreaks(pContent);
847
+ for (let k = 0; k < segments.length; k++) {
848
+ const segment = segments[k];
849
+ if (!segment.text.trim() && segment.children.length === 0)
850
+ continue;
851
+ const isFirst = k === 0;
852
+ const nodeType = isFirst ? 'list' : 'paragraph';
853
+ const node = {
854
+ type: nodeType,
855
+ text: segment.text,
856
+ children: segment.children,
857
+ metadata: isFirst ? {
858
+ listType,
859
+ indentation,
860
+ itemIndex,
861
+ listId,
862
+ alignment: pContent.alignment || 'left',
863
+ style: pContent.style
864
+ } : {
865
+ alignment: pContent.alignment || 'left',
866
+ style: pContent.style
867
+ }
868
+ };
869
+ // Special case for headings in lists
870
+ if (isFirst && element.tagName === "text:h") {
871
+ const level = parseInt(element.getAttribute("text:outline-level") || "1");
872
+ node.metadata.level = level;
825
873
  }
826
- };
827
- if (config.includeRawContent)
828
- listNode.rawContent = (0, xmlUtils_js_1.getRawContent)(element, sourceXml, config);
829
- targetArray.push(listNode);
874
+ if (config.includeRawContent)
875
+ node.rawContent = (0, xmlUtils_js_1.getRawContent)(element, sourceXml, config);
876
+ targetArray.push(node);
877
+ }
830
878
  }
831
879
  else if (element.tagName === "text:list") {
832
880
  // Recursive call for nested list
@@ -392,117 +392,129 @@ const parsePowerPoint = async (buffer, config) => {
392
392
  if (config.includeRawContent) {
393
393
  pNode.rawContent = (0, xmlUtils_js_1.getRawContent)(p, xmlContentString, config);
394
394
  }
395
- const runs = (0, xmlUtils_js_1.getElementsByTagName)(p, "a:r");
396
- for (let j = 0; j < runs.length; j++) {
397
- const r = runs[j];
398
- const t = (0, xmlUtils_js_1.getFirstElementByTagName)(r, "a:t");
399
- if (t && t.childNodes[0]) {
400
- const textContent = t.childNodes[0].nodeValue || '';
401
- pNode.text += textContent;
402
- const rPr = (0, xmlUtils_js_1.getFirstElementByTagName)(r, "a:rPr");
403
- const formatting = {};
404
- if (rPr) {
405
- if (rPr.getAttribute("b") === "1")
406
- formatting.bold = true;
407
- if (rPr.getAttribute("i") === "1")
408
- formatting.italic = true;
409
- if (rPr.getAttribute("u") === "sng")
410
- formatting.underline = true;
411
- if (rPr.getAttribute("strike") === "sngStrike")
412
- formatting.strikethrough = true;
413
- const sz = rPr.getAttribute("sz");
414
- if (sz)
415
- formatting.size = (parseInt(sz) / 100).toString() + 'pt';
416
- // Color extraction
417
- const solidFill = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:solidFill");
418
- if (solidFill) {
419
- const srgbClr = (0, xmlUtils_js_1.getFirstElementByTagName)(solidFill, "a:srgbClr");
420
- if (srgbClr) {
421
- const val = srgbClr.getAttribute("val");
422
- if (val)
423
- formatting.color = '#' + val;
395
+ // Process all children of <a:p> in order (runs, breaks, fields)
396
+ const children = Array.from(p.childNodes);
397
+ let activeNode = pNode;
398
+ nodes.push(activeNode);
399
+ for (const childNode of children) {
400
+ if (!(0, xmlUtils_js_1.isElement)(childNode))
401
+ continue;
402
+ const element = childNode;
403
+ const tag = element.tagName;
404
+ if (tag === "a:r" || tag === "a:fld") {
405
+ const t = (0, xmlUtils_js_1.getFirstElementByTagName)(element, "a:t");
406
+ if (t && t.childNodes[0]) {
407
+ const textContent = t.childNodes[0].nodeValue || "";
408
+ activeNode.text += textContent;
409
+ const rPr = (0, xmlUtils_js_1.getFirstElementByTagName)(element, "a:rPr");
410
+ const formatting = {};
411
+ if (rPr) {
412
+ if (rPr.getAttribute("b") === "1")
413
+ formatting.bold = true;
414
+ if (rPr.getAttribute("i") === "1")
415
+ formatting.italic = true;
416
+ if (rPr.getAttribute("u") === "sng")
417
+ formatting.underline = true;
418
+ if (rPr.getAttribute("strike") === "sngStrike")
419
+ formatting.strikethrough = true;
420
+ const sz = rPr.getAttribute("sz");
421
+ if (sz)
422
+ formatting.size = (parseInt(sz) / 100).toString() + "pt";
423
+ // Color extraction
424
+ const solidFill = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:solidFill");
425
+ if (solidFill) {
426
+ const srgbClr = (0, xmlUtils_js_1.getFirstElementByTagName)(solidFill, "a:srgbClr");
427
+ if (srgbClr) {
428
+ const val = srgbClr.getAttribute("val");
429
+ if (val)
430
+ formatting.color = "#" + val;
431
+ }
424
432
  }
425
- }
426
- // Highlight extraction
427
- const highlight = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:highlight");
428
- if (highlight) {
429
- const srgbClr = (0, xmlUtils_js_1.getFirstElementByTagName)(highlight, "a:srgbClr");
430
- if (srgbClr) {
431
- const val = srgbClr.getAttribute("val");
432
- if (val)
433
- formatting.backgroundColor = '#' + val;
433
+ // Highlight extraction
434
+ const highlight = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:highlight");
435
+ if (highlight) {
436
+ const srgbClr = (0, xmlUtils_js_1.getFirstElementByTagName)(highlight, "a:srgbClr");
437
+ if (srgbClr) {
438
+ const val = srgbClr.getAttribute("val");
439
+ if (val)
440
+ formatting.backgroundColor = "#" + val;
441
+ }
442
+ }
443
+ // Font family
444
+ const latin = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:latin");
445
+ if (latin) {
446
+ const typeface = latin.getAttribute("typeface");
447
+ if (typeface)
448
+ formatting.font = typeface;
449
+ }
450
+ // Subscript/Superscript
451
+ const baseline = rPr.getAttribute("baseline");
452
+ if (baseline) {
453
+ const baselineVal = parseInt(baseline);
454
+ if (baselineVal < 0)
455
+ formatting.subscript = true;
456
+ if (baselineVal > 0)
457
+ formatting.superscript = true;
434
458
  }
435
459
  }
436
- // Font family
437
- const latin = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:latin");
438
- if (latin) {
439
- const typeface = latin.getAttribute("typeface");
440
- if (typeface)
441
- formatting.font = typeface;
442
- }
443
- // Subscript/Superscript
444
- const baseline = rPr.getAttribute("baseline");
445
- if (baseline) {
446
- const baselineVal = parseInt(baseline);
447
- if (baselineVal < 0)
448
- formatting.subscript = true;
449
- if (baselineVal > 0)
450
- formatting.superscript = true;
460
+ const textNode = {
461
+ type: 'text',
462
+ text: textContent,
463
+ formatting: formatting
464
+ };
465
+ // Check for Hyperlinks
466
+ const hlinkClick = (0, xmlUtils_js_1.getFirstElementByTagName)(element, "a:hlinkClick");
467
+ if (hlinkClick) {
468
+ const rId = hlinkClick.getAttribute("r:id");
469
+ const action = hlinkClick.getAttribute("action");
470
+ let link;
471
+ let linkType;
472
+ if (rId && slideRelsMap[slideNumber] && slideRelsMap[slideNumber][rId] && slideRelsMap[slideNumber][rId].type === "hyperlink") {
473
+ link = slideRelsMap[slideNumber][rId].target;
474
+ linkType = "external";
475
+ }
476
+ else if (rId && slideRelsMap[slideNumber] && slideRelsMap[slideNumber][rId] && slideRelsMap[slideNumber][rId].type === "slide") {
477
+ link = slideRelsMap[slideNumber][rId].target;
478
+ linkType = "internal";
479
+ }
480
+ else if (action) {
481
+ link = action;
482
+ linkType = "internal";
483
+ }
484
+ if (link) {
485
+ textNode.metadata = { link, linkType };
486
+ }
451
487
  }
488
+ activeNode.children?.push(textNode);
452
489
  }
453
- const textNode = {
454
- type: 'text',
455
- text: textContent,
456
- formatting: formatting
457
- };
458
- // Check for Hyperlinks
459
- const hlinkClick = (0, xmlUtils_js_1.getFirstElementByTagName)(r, "a:hlinkClick");
460
- // Check if this run has a hyperlink click action
461
- if (hlinkClick) {
462
- // Relationship ID for the link
463
- const rId = hlinkClick.getAttribute("r:id");
464
- // Optional action attribute, often for internal jumps
465
- const action = hlinkClick.getAttribute("action");
466
- // Result placeholders
467
- let link;
468
- let linkType;
469
- // Case 1: Relationship exists in slideRelsMap and is a real hyperlink (external URL)
470
- if (rId
471
- && slideRelsMap[slideNumber]
472
- && slideRelsMap[slideNumber][rId]
473
- && slideRelsMap[slideNumber][rId].type === "hyperlink") {
474
- // External URL
475
- link = slideRelsMap[slideNumber][rId].target;
476
- linkType = "external";
477
- }
478
- // Case 2: Relationship exists and is an internal slide reference
479
- else if (rId
480
- && slideRelsMap[slideNumber]
481
- && slideRelsMap[slideNumber][rId]
482
- && slideRelsMap[slideNumber][rId].type === "slide") {
483
- // Example target: ppt/slides/slide3.xml
484
- link = slideRelsMap[slideNumber][rId].target;
485
- linkType = "internal";
486
- }
487
- // Case 3: action attribute like ppaction://hlinksldjump
488
- else if (action) {
489
- link = action;
490
- linkType = "internal";
491
- }
492
- // Assign metadata only if a link was actually discovered
493
- if (link) {
494
- textNode.metadata = { link, linkType };
490
+ }
491
+ else if (tag === "a:br") {
492
+ if (isList) {
493
+ // Split the list item on soft break into a paragraph node
494
+ activeNode = {
495
+ type: 'paragraph',
496
+ text: '',
497
+ children: [],
498
+ metadata: {
499
+ indentation: lvl,
500
+ alignment: pNode.metadata?.alignment || 'left'
501
+ }
502
+ };
503
+ if (config.includeRawContent) {
504
+ activeNode.rawContent = (0, xmlUtils_js_1.getRawContent)(p, xmlContentString, config);
495
505
  }
506
+ nodes.push(activeNode);
507
+ }
508
+ else {
509
+ // In a normal paragraph, just add a newline
510
+ activeNode.text += "\n";
511
+ activeNode.children?.push({ type: 'text', text: "\n" });
496
512
  }
497
- pNode.children?.push(textNode);
498
513
  }
499
514
  }
500
- if (pNode.text) {
501
- nodes.push(pNode);
502
- }
503
515
  }
504
516
  }
505
- return nodes;
517
+ return nodes.filter(n => n.text?.trim() || (n.children && n.children.length > 0));
506
518
  };
507
519
  /**
508
520
  * Recursively traverses a PowerPoint shape tree (p:spTree),
@@ -130,6 +130,12 @@ export declare class SimpleRtfParser {
130
130
  private index;
131
131
  /** The RTF content as a Buffer */
132
132
  private buffer;
133
+ /** Current code page for character decoding (default is Windows-1252) */
134
+ private codePage;
135
+ /** Cached TextDecoders for different code pages */
136
+ private decoders;
137
+ /** Buffer for consecutive text bytes to handle multi-byte encodings and UTF-8 detection */
138
+ private pendingBytes;
133
139
  /** Total length of the buffer */
134
140
  private length;
135
141
  /**
@@ -140,6 +146,20 @@ export declare class SimpleRtfParser {
140
146
  parse(): RtfGroup;
141
147
  private parseControl;
142
148
  private parseText;
149
+ /**
150
+ * Flushes the pending bytes buffer as a text node to the current group.
151
+ * @param group The group to append the text node to
152
+ */
153
+ private flushPendingText;
154
+ /**
155
+ * Decodes a byte array using a "UTF-8 first" strategy.
156
+ * If the bytes form valid UTF-8 and contain non-ASCII characters, UTF-8 is preferred.
157
+ * Otherwise, falls back to the specified code page.
158
+ * @param bytes The bytes to decode
159
+ * @param codePage The RTF code page ID
160
+ * @returns The decoded string
161
+ */
162
+ private decodeBytes;
143
163
  }
144
164
  /**
145
165
  * Parses an RTF file and returns the AST.