officeparser 6.1.0 → 6.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +68 -63
- package/dist/OfficeParser.js +1 -0
- package/dist/cli.js +1 -0
- package/dist/index.d.ts +2 -2
- package/dist/officeparser.browser.d.ts +54 -2
- package/dist/officeparser.browser.iife.js +49 -46
- package/dist/officeparser.browser.mjs +49 -46
- package/dist/parsers/ExcelParser.js +5 -5
- package/dist/parsers/OpenOfficeParser.js +97 -49
- package/dist/parsers/PowerPointParser.js +112 -100
- package/dist/parsers/RtfParser.d.ts +20 -0
- package/dist/parsers/RtfParser.js +107 -42
- package/dist/parsers/WordParser.d.ts +1 -0
- package/dist/parsers/WordParser.js +107 -24
- package/dist/sbom.cdx.json +102 -102
- package/dist/types.d.ts +54 -2
- package/package.json +2 -2
|
@@ -422,10 +422,10 @@ const parseExcel = async (buffer, config) => {
|
|
|
422
422
|
if (cMatches) {
|
|
423
423
|
for (const cXml of cMatches) {
|
|
424
424
|
// Extract cell value
|
|
425
|
-
const typeMatch = cXml.match(/t="([a-
|
|
425
|
+
const typeMatch = cXml.match(/t="([a-zA-Z]+)"/);
|
|
426
426
|
const type = typeMatch ? typeMatch[1] : 'n'; // n = number (default)
|
|
427
|
-
const vMatch = cXml.match(/<v>(
|
|
428
|
-
const tMatch = cXml.match(/<t>(
|
|
427
|
+
const vMatch = cXml.match(/<v>([\s\S]*?)<\/v>/);
|
|
428
|
+
const tMatch = cXml.match(/<t>([\s\S]*?)<\/t>/);
|
|
429
429
|
let text = '';
|
|
430
430
|
let cellNodes = [];
|
|
431
431
|
if (type === 's' && vMatch) {
|
|
@@ -442,10 +442,10 @@ const parseExcel = async (buffer, config) => {
|
|
|
442
442
|
}
|
|
443
443
|
}
|
|
444
444
|
else if (type === 'inlineStr' && tMatch) {
|
|
445
|
-
text = tMatch[1];
|
|
445
|
+
text = tMatch[1].trim();
|
|
446
446
|
}
|
|
447
447
|
else if (vMatch) {
|
|
448
|
-
text = vMatch[1];
|
|
448
|
+
text = vMatch[1].trim();
|
|
449
449
|
}
|
|
450
450
|
// Parse cell coordinate
|
|
451
451
|
const coordMatch = cXml.match(/r="([A-Z]+)(\d+)"/);
|
|
@@ -70,6 +70,11 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
70
70
|
const styleMap = {};
|
|
71
71
|
const paragraphStyleMap = {};
|
|
72
72
|
const listCounters = {}; // Track item index per listId/level
|
|
73
|
+
let currentListId = null;
|
|
74
|
+
let lastListType = null;
|
|
75
|
+
let lastListStyle = null;
|
|
76
|
+
let listIdCounter = 0;
|
|
77
|
+
let lastWasList = false;
|
|
73
78
|
// Helper to parse styles
|
|
74
79
|
const parseStyles = (xmlString) => {
|
|
75
80
|
const xml = (0, xmlUtils_js_1.parseXmlString)(xmlString);
|
|
@@ -221,7 +226,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
221
226
|
type: 'text',
|
|
222
227
|
text: '\n',
|
|
223
228
|
formatting: parentFormatting,
|
|
224
|
-
metadata:
|
|
229
|
+
metadata: { ...(linkMetadata || {}), isLineBreak: true }
|
|
225
230
|
});
|
|
226
231
|
}
|
|
227
232
|
else if (tagName === 'text:span') {
|
|
@@ -391,6 +396,31 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
391
396
|
}
|
|
392
397
|
return { text: content.text, children: content.children, alignment, style: paraStyle || undefined };
|
|
393
398
|
};
|
|
399
|
+
/**
|
|
400
|
+
* Splits paragraph content into multiple segments based on line breaks.
|
|
401
|
+
* Used to handle soft line breaks within list items.
|
|
402
|
+
*
|
|
403
|
+
* @param pContent - The content of a single paragraph
|
|
404
|
+
* @returns Array of content segments
|
|
405
|
+
*/
|
|
406
|
+
const splitParagraphByBreaks = (pContent) => {
|
|
407
|
+
const segments = [];
|
|
408
|
+
let currentText = "";
|
|
409
|
+
let currentChildren = [];
|
|
410
|
+
for (const child of pContent.children) {
|
|
411
|
+
if (child.type === "text" && child.metadata?.isLineBreak) {
|
|
412
|
+
segments.push({ text: currentText, children: currentChildren });
|
|
413
|
+
currentText = "";
|
|
414
|
+
currentChildren = [];
|
|
415
|
+
}
|
|
416
|
+
else {
|
|
417
|
+
currentText += child.text || "";
|
|
418
|
+
currentChildren.push(child);
|
|
419
|
+
}
|
|
420
|
+
}
|
|
421
|
+
segments.push({ text: currentText, children: currentChildren });
|
|
422
|
+
return segments;
|
|
423
|
+
};
|
|
394
424
|
/**
|
|
395
425
|
* Helper to parse a table node and extract its structure.
|
|
396
426
|
* Properly creates table → row → cell hierarchy with metadata.
|
|
@@ -640,6 +670,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
640
670
|
pNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
|
|
641
671
|
}
|
|
642
672
|
targetArray.push(pNode);
|
|
673
|
+
lastWasList = false;
|
|
643
674
|
}
|
|
644
675
|
else if (node.tagName === "text:h") {
|
|
645
676
|
const level = parseInt(node.getAttribute("text:outline-level") || "1");
|
|
@@ -658,6 +689,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
658
689
|
hNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
|
|
659
690
|
}
|
|
660
691
|
targetArray.push(hNode);
|
|
692
|
+
lastWasList = false;
|
|
661
693
|
}
|
|
662
694
|
else if (node.tagName === "table:table") {
|
|
663
695
|
// Parse table with proper structure
|
|
@@ -666,16 +698,15 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
666
698
|
tableNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
|
|
667
699
|
}
|
|
668
700
|
targetArray.push(tableNode);
|
|
701
|
+
lastWasList = false;
|
|
669
702
|
}
|
|
670
703
|
else if (node.tagName === "text:list") {
|
|
671
704
|
// Parse list structure with proper listId tracking
|
|
672
705
|
const listItems = (0, xmlUtils_js_1.getDirectChildren)(node, "text:list-item");
|
|
673
|
-
// Get list style name to use as listId (or generate one)
|
|
674
|
-
const listStyleName = node.getAttribute("text:style-name") || node.getAttribute("xml:id");
|
|
675
|
-
const listId = listStyleName || `list-${targetArray.length}`;
|
|
676
706
|
// Determine list type by checking the list style definition
|
|
677
707
|
let listType = 'unordered';
|
|
678
708
|
let isVisible = false;
|
|
709
|
+
const listStyleName = node.getAttribute("text:style-name") || node.getAttribute("xml:id");
|
|
679
710
|
let styleNameToCheck = listStyleName;
|
|
680
711
|
// If no style name, check parent list for inherited style
|
|
681
712
|
if (!styleNameToCheck) {
|
|
@@ -741,6 +772,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
741
772
|
// If the list is not visible, it's likely a layout list used by Impress.
|
|
742
773
|
// We should traverse its items and treat their content as regular nodes.
|
|
743
774
|
if (!isVisible) {
|
|
775
|
+
lastWasList = false;
|
|
744
776
|
for (let i = 0; i < listItems.length; i++) {
|
|
745
777
|
const item = listItems[i];
|
|
746
778
|
if (item.childNodes) {
|
|
@@ -754,6 +786,24 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
754
786
|
}
|
|
755
787
|
return;
|
|
756
788
|
}
|
|
789
|
+
// List Continuity Logic:
|
|
790
|
+
// If this list follows another list of the same type and style, or we are in ODP and it's sequential,
|
|
791
|
+
// we should reuse the previous listId to maintain numbering.
|
|
792
|
+
const isODP = fileType === 'odp';
|
|
793
|
+
const sameStyle = styleNameToCheck && styleNameToCheck === lastListStyle;
|
|
794
|
+
const sameType = listType === lastListType;
|
|
795
|
+
let listId;
|
|
796
|
+
if (lastWasList && (sameStyle || (isODP && sameType))) {
|
|
797
|
+
listId = currentListId;
|
|
798
|
+
}
|
|
799
|
+
else {
|
|
800
|
+
// New list
|
|
801
|
+
listId = styleNameToCheck || `list-${++listIdCounter}`;
|
|
802
|
+
currentListId = listId;
|
|
803
|
+
lastListType = listType;
|
|
804
|
+
lastListStyle = styleNameToCheck;
|
|
805
|
+
}
|
|
806
|
+
lastWasList = true;
|
|
757
807
|
// Calculate indentation level by counting parent text:list elements
|
|
758
808
|
let indentation = 0;
|
|
759
809
|
let parent = node.parentNode;
|
|
@@ -774,59 +824,57 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
774
824
|
// Process each list item
|
|
775
825
|
for (let i = 0; i < listItems.length; i++) {
|
|
776
826
|
const item = listItems[i];
|
|
777
|
-
|
|
778
|
-
listCounters[listId][indentKey]++;
|
|
779
|
-
const itemIndex = listCounters[listId][indentKey];
|
|
780
|
-
// Reset deeper levels when we encounter an item at this level
|
|
781
|
-
for (let k = indentation + 1; k < 10; k++) {
|
|
782
|
-
if (listCounters[listId][k.toString()] !== undefined) {
|
|
783
|
-
listCounters[listId][k.toString()] = -1;
|
|
784
|
-
}
|
|
785
|
-
}
|
|
827
|
+
let hasIndexedThisItem = false;
|
|
786
828
|
// Iterate over direct children of list item (paragraphs, headings, nested lists)
|
|
787
829
|
if (item.childNodes) {
|
|
788
830
|
for (let j = 0; j < item.childNodes.length; j++) {
|
|
789
831
|
const child = item.childNodes[j];
|
|
790
832
|
if ((0, xmlUtils_js_1.isElement)(child)) { // Element
|
|
791
833
|
const element = child;
|
|
792
|
-
if (element.tagName === "text:p") {
|
|
793
|
-
|
|
794
|
-
|
|
795
|
-
|
|
796
|
-
|
|
797
|
-
|
|
798
|
-
|
|
799
|
-
|
|
800
|
-
indentation,
|
|
801
|
-
itemIndex,
|
|
802
|
-
listId,
|
|
803
|
-
alignment: pContent.alignment || 'left',
|
|
804
|
-
style: pContent.style
|
|
834
|
+
if (element.tagName === "text:p" || element.tagName === "text:h") {
|
|
835
|
+
if (!hasIndexedThisItem) {
|
|
836
|
+
listCounters[listId][indentKey]++;
|
|
837
|
+
hasIndexedThisItem = true;
|
|
838
|
+
for (let k = indentation + 1; k < 10; k++) {
|
|
839
|
+
if (listCounters[listId][k.toString()] !== undefined) {
|
|
840
|
+
listCounters[listId][k.toString()] = -1;
|
|
841
|
+
}
|
|
805
842
|
}
|
|
806
|
-
}
|
|
807
|
-
|
|
808
|
-
|
|
809
|
-
|
|
810
|
-
|
|
811
|
-
|
|
812
|
-
|
|
813
|
-
|
|
814
|
-
|
|
815
|
-
|
|
816
|
-
|
|
817
|
-
|
|
818
|
-
|
|
819
|
-
|
|
820
|
-
|
|
821
|
-
|
|
822
|
-
|
|
823
|
-
|
|
824
|
-
|
|
843
|
+
}
|
|
844
|
+
const itemIndex = listCounters[listId][indentKey];
|
|
845
|
+
const pContent = parseParagraphContent(element, paragraphStyleMap, styleMap, config, sourceXml);
|
|
846
|
+
const segments = splitParagraphByBreaks(pContent);
|
|
847
|
+
for (let k = 0; k < segments.length; k++) {
|
|
848
|
+
const segment = segments[k];
|
|
849
|
+
if (!segment.text.trim() && segment.children.length === 0)
|
|
850
|
+
continue;
|
|
851
|
+
const isFirst = k === 0;
|
|
852
|
+
const nodeType = isFirst ? 'list' : 'paragraph';
|
|
853
|
+
const node = {
|
|
854
|
+
type: nodeType,
|
|
855
|
+
text: segment.text,
|
|
856
|
+
children: segment.children,
|
|
857
|
+
metadata: isFirst ? {
|
|
858
|
+
listType,
|
|
859
|
+
indentation,
|
|
860
|
+
itemIndex,
|
|
861
|
+
listId,
|
|
862
|
+
alignment: pContent.alignment || 'left',
|
|
863
|
+
style: pContent.style
|
|
864
|
+
} : {
|
|
865
|
+
alignment: pContent.alignment || 'left',
|
|
866
|
+
style: pContent.style
|
|
867
|
+
}
|
|
868
|
+
};
|
|
869
|
+
// Special case for headings in lists
|
|
870
|
+
if (isFirst && element.tagName === "text:h") {
|
|
871
|
+
const level = parseInt(element.getAttribute("text:outline-level") || "1");
|
|
872
|
+
node.metadata.level = level;
|
|
825
873
|
}
|
|
826
|
-
|
|
827
|
-
|
|
828
|
-
|
|
829
|
-
|
|
874
|
+
if (config.includeRawContent)
|
|
875
|
+
node.rawContent = (0, xmlUtils_js_1.getRawContent)(element, sourceXml, config);
|
|
876
|
+
targetArray.push(node);
|
|
877
|
+
}
|
|
830
878
|
}
|
|
831
879
|
else if (element.tagName === "text:list") {
|
|
832
880
|
// Recursive call for nested list
|
|
@@ -392,117 +392,129 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
392
392
|
if (config.includeRawContent) {
|
|
393
393
|
pNode.rawContent = (0, xmlUtils_js_1.getRawContent)(p, xmlContentString, config);
|
|
394
394
|
}
|
|
395
|
-
|
|
396
|
-
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
|
|
404
|
-
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
if (rPr
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
|
|
421
|
-
|
|
422
|
-
|
|
423
|
-
|
|
395
|
+
// Process all children of <a:p> in order (runs, breaks, fields)
|
|
396
|
+
const children = Array.from(p.childNodes);
|
|
397
|
+
let activeNode = pNode;
|
|
398
|
+
nodes.push(activeNode);
|
|
399
|
+
for (const childNode of children) {
|
|
400
|
+
if (!(0, xmlUtils_js_1.isElement)(childNode))
|
|
401
|
+
continue;
|
|
402
|
+
const element = childNode;
|
|
403
|
+
const tag = element.tagName;
|
|
404
|
+
if (tag === "a:r" || tag === "a:fld") {
|
|
405
|
+
const t = (0, xmlUtils_js_1.getFirstElementByTagName)(element, "a:t");
|
|
406
|
+
if (t && t.childNodes[0]) {
|
|
407
|
+
const textContent = t.childNodes[0].nodeValue || "";
|
|
408
|
+
activeNode.text += textContent;
|
|
409
|
+
const rPr = (0, xmlUtils_js_1.getFirstElementByTagName)(element, "a:rPr");
|
|
410
|
+
const formatting = {};
|
|
411
|
+
if (rPr) {
|
|
412
|
+
if (rPr.getAttribute("b") === "1")
|
|
413
|
+
formatting.bold = true;
|
|
414
|
+
if (rPr.getAttribute("i") === "1")
|
|
415
|
+
formatting.italic = true;
|
|
416
|
+
if (rPr.getAttribute("u") === "sng")
|
|
417
|
+
formatting.underline = true;
|
|
418
|
+
if (rPr.getAttribute("strike") === "sngStrike")
|
|
419
|
+
formatting.strikethrough = true;
|
|
420
|
+
const sz = rPr.getAttribute("sz");
|
|
421
|
+
if (sz)
|
|
422
|
+
formatting.size = (parseInt(sz) / 100).toString() + "pt";
|
|
423
|
+
// Color extraction
|
|
424
|
+
const solidFill = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:solidFill");
|
|
425
|
+
if (solidFill) {
|
|
426
|
+
const srgbClr = (0, xmlUtils_js_1.getFirstElementByTagName)(solidFill, "a:srgbClr");
|
|
427
|
+
if (srgbClr) {
|
|
428
|
+
const val = srgbClr.getAttribute("val");
|
|
429
|
+
if (val)
|
|
430
|
+
formatting.color = "#" + val;
|
|
431
|
+
}
|
|
424
432
|
}
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
|
|
429
|
-
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
|
|
433
|
-
|
|
433
|
+
// Highlight extraction
|
|
434
|
+
const highlight = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:highlight");
|
|
435
|
+
if (highlight) {
|
|
436
|
+
const srgbClr = (0, xmlUtils_js_1.getFirstElementByTagName)(highlight, "a:srgbClr");
|
|
437
|
+
if (srgbClr) {
|
|
438
|
+
const val = srgbClr.getAttribute("val");
|
|
439
|
+
if (val)
|
|
440
|
+
formatting.backgroundColor = "#" + val;
|
|
441
|
+
}
|
|
442
|
+
}
|
|
443
|
+
// Font family
|
|
444
|
+
const latin = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:latin");
|
|
445
|
+
if (latin) {
|
|
446
|
+
const typeface = latin.getAttribute("typeface");
|
|
447
|
+
if (typeface)
|
|
448
|
+
formatting.font = typeface;
|
|
449
|
+
}
|
|
450
|
+
// Subscript/Superscript
|
|
451
|
+
const baseline = rPr.getAttribute("baseline");
|
|
452
|
+
if (baseline) {
|
|
453
|
+
const baselineVal = parseInt(baseline);
|
|
454
|
+
if (baselineVal < 0)
|
|
455
|
+
formatting.subscript = true;
|
|
456
|
+
if (baselineVal > 0)
|
|
457
|
+
formatting.superscript = true;
|
|
434
458
|
}
|
|
435
459
|
}
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
|
|
441
|
-
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
|
|
446
|
-
|
|
447
|
-
|
|
448
|
-
|
|
449
|
-
|
|
450
|
-
|
|
460
|
+
const textNode = {
|
|
461
|
+
type: 'text',
|
|
462
|
+
text: textContent,
|
|
463
|
+
formatting: formatting
|
|
464
|
+
};
|
|
465
|
+
// Check for Hyperlinks
|
|
466
|
+
const hlinkClick = (0, xmlUtils_js_1.getFirstElementByTagName)(element, "a:hlinkClick");
|
|
467
|
+
if (hlinkClick) {
|
|
468
|
+
const rId = hlinkClick.getAttribute("r:id");
|
|
469
|
+
const action = hlinkClick.getAttribute("action");
|
|
470
|
+
let link;
|
|
471
|
+
let linkType;
|
|
472
|
+
if (rId && slideRelsMap[slideNumber] && slideRelsMap[slideNumber][rId] && slideRelsMap[slideNumber][rId].type === "hyperlink") {
|
|
473
|
+
link = slideRelsMap[slideNumber][rId].target;
|
|
474
|
+
linkType = "external";
|
|
475
|
+
}
|
|
476
|
+
else if (rId && slideRelsMap[slideNumber] && slideRelsMap[slideNumber][rId] && slideRelsMap[slideNumber][rId].type === "slide") {
|
|
477
|
+
link = slideRelsMap[slideNumber][rId].target;
|
|
478
|
+
linkType = "internal";
|
|
479
|
+
}
|
|
480
|
+
else if (action) {
|
|
481
|
+
link = action;
|
|
482
|
+
linkType = "internal";
|
|
483
|
+
}
|
|
484
|
+
if (link) {
|
|
485
|
+
textNode.metadata = { link, linkType };
|
|
486
|
+
}
|
|
451
487
|
}
|
|
488
|
+
activeNode.children?.push(textNode);
|
|
452
489
|
}
|
|
453
|
-
|
|
454
|
-
|
|
455
|
-
|
|
456
|
-
|
|
457
|
-
|
|
458
|
-
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
|
|
462
|
-
|
|
463
|
-
|
|
464
|
-
|
|
465
|
-
|
|
466
|
-
|
|
467
|
-
|
|
468
|
-
let linkType;
|
|
469
|
-
// Case 1: Relationship exists in slideRelsMap and is a real hyperlink (external URL)
|
|
470
|
-
if (rId
|
|
471
|
-
&& slideRelsMap[slideNumber]
|
|
472
|
-
&& slideRelsMap[slideNumber][rId]
|
|
473
|
-
&& slideRelsMap[slideNumber][rId].type === "hyperlink") {
|
|
474
|
-
// External URL
|
|
475
|
-
link = slideRelsMap[slideNumber][rId].target;
|
|
476
|
-
linkType = "external";
|
|
477
|
-
}
|
|
478
|
-
// Case 2: Relationship exists and is an internal slide reference
|
|
479
|
-
else if (rId
|
|
480
|
-
&& slideRelsMap[slideNumber]
|
|
481
|
-
&& slideRelsMap[slideNumber][rId]
|
|
482
|
-
&& slideRelsMap[slideNumber][rId].type === "slide") {
|
|
483
|
-
// Example target: ppt/slides/slide3.xml
|
|
484
|
-
link = slideRelsMap[slideNumber][rId].target;
|
|
485
|
-
linkType = "internal";
|
|
486
|
-
}
|
|
487
|
-
// Case 3: action attribute like ppaction://hlinksldjump
|
|
488
|
-
else if (action) {
|
|
489
|
-
link = action;
|
|
490
|
-
linkType = "internal";
|
|
491
|
-
}
|
|
492
|
-
// Assign metadata only if a link was actually discovered
|
|
493
|
-
if (link) {
|
|
494
|
-
textNode.metadata = { link, linkType };
|
|
490
|
+
}
|
|
491
|
+
else if (tag === "a:br") {
|
|
492
|
+
if (isList) {
|
|
493
|
+
// Split the list item on soft break into a paragraph node
|
|
494
|
+
activeNode = {
|
|
495
|
+
type: 'paragraph',
|
|
496
|
+
text: '',
|
|
497
|
+
children: [],
|
|
498
|
+
metadata: {
|
|
499
|
+
indentation: lvl,
|
|
500
|
+
alignment: pNode.metadata?.alignment || 'left'
|
|
501
|
+
}
|
|
502
|
+
};
|
|
503
|
+
if (config.includeRawContent) {
|
|
504
|
+
activeNode.rawContent = (0, xmlUtils_js_1.getRawContent)(p, xmlContentString, config);
|
|
495
505
|
}
|
|
506
|
+
nodes.push(activeNode);
|
|
507
|
+
}
|
|
508
|
+
else {
|
|
509
|
+
// In a normal paragraph, just add a newline
|
|
510
|
+
activeNode.text += "\n";
|
|
511
|
+
activeNode.children?.push({ type: 'text', text: "\n" });
|
|
496
512
|
}
|
|
497
|
-
pNode.children?.push(textNode);
|
|
498
513
|
}
|
|
499
514
|
}
|
|
500
|
-
if (pNode.text) {
|
|
501
|
-
nodes.push(pNode);
|
|
502
|
-
}
|
|
503
515
|
}
|
|
504
516
|
}
|
|
505
|
-
return nodes;
|
|
517
|
+
return nodes.filter(n => n.text?.trim() || (n.children && n.children.length > 0));
|
|
506
518
|
};
|
|
507
519
|
/**
|
|
508
520
|
* Recursively traverses a PowerPoint shape tree (p:spTree),
|
|
@@ -130,6 +130,12 @@ export declare class SimpleRtfParser {
|
|
|
130
130
|
private index;
|
|
131
131
|
/** The RTF content as a Buffer */
|
|
132
132
|
private buffer;
|
|
133
|
+
/** Current code page for character decoding (default is Windows-1252) */
|
|
134
|
+
private codePage;
|
|
135
|
+
/** Cached TextDecoders for different code pages */
|
|
136
|
+
private decoders;
|
|
137
|
+
/** Buffer for consecutive text bytes to handle multi-byte encodings and UTF-8 detection */
|
|
138
|
+
private pendingBytes;
|
|
133
139
|
/** Total length of the buffer */
|
|
134
140
|
private length;
|
|
135
141
|
/**
|
|
@@ -140,6 +146,20 @@ export declare class SimpleRtfParser {
|
|
|
140
146
|
parse(): RtfGroup;
|
|
141
147
|
private parseControl;
|
|
142
148
|
private parseText;
|
|
149
|
+
/**
|
|
150
|
+
* Flushes the pending bytes buffer as a text node to the current group.
|
|
151
|
+
* @param group The group to append the text node to
|
|
152
|
+
*/
|
|
153
|
+
private flushPendingText;
|
|
154
|
+
/**
|
|
155
|
+
* Decodes a byte array using a "UTF-8 first" strategy.
|
|
156
|
+
* If the bytes form valid UTF-8 and contain non-ASCII characters, UTF-8 is preferred.
|
|
157
|
+
* Otherwise, falls back to the specified code page.
|
|
158
|
+
* @param bytes The bytes to decode
|
|
159
|
+
* @param codePage The RTF code page ID
|
|
160
|
+
* @returns The decoded string
|
|
161
|
+
*/
|
|
162
|
+
private decodeBytes;
|
|
143
163
|
}
|
|
144
164
|
/**
|
|
145
165
|
* Parses an RTF file and returns the AST.
|