officeparser 6.1.0 → 7.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +284 -86
- package/dist/OfficeConverter.d.ts +46 -0
- package/dist/OfficeConverter.js +72 -0
- package/dist/OfficeGenerator.d.ts +19 -0
- package/dist/OfficeGenerator.js +48 -0
- package/dist/OfficeParser.d.ts +6 -0
- package/dist/OfficeParser.js +55 -28
- package/dist/cli.d.ts +3 -1
- package/dist/cli.js +107 -22
- package/dist/defaults.d.ts +41 -0
- package/dist/defaults.js +172 -0
- package/dist/generators/BaseGenerator.d.ts +58 -0
- package/dist/generators/BaseGenerator.js +107 -0
- package/dist/generators/ChunkingGenerator.d.ts +81 -0
- package/dist/generators/ChunkingGenerator.js +683 -0
- package/dist/generators/CsvGenerator.d.ts +30 -0
- package/dist/generators/CsvGenerator.js +233 -0
- package/dist/generators/HtmlGenerator.d.ts +37 -0
- package/dist/generators/HtmlGenerator.js +1013 -0
- package/dist/generators/MarkdownGenerator.d.ts +59 -0
- package/dist/generators/MarkdownGenerator.js +481 -0
- package/dist/generators/PdfGenerator.d.ts +22 -0
- package/dist/generators/PdfGenerator.js +118 -0
- package/dist/generators/RtfGenerator.d.ts +15 -0
- package/dist/generators/RtfGenerator.js +208 -0
- package/dist/generators/TextGenerator.d.ts +13 -0
- package/dist/generators/TextGenerator.js +108 -0
- package/dist/index.d.ts +11 -3
- package/dist/index.js +17 -2
- package/dist/index.mjs +2 -2
- package/dist/officeparser.browser.d.ts +878 -5
- package/dist/officeparser.browser.iife.js +703 -49
- package/dist/officeparser.browser.mjs +703 -49
- package/dist/parsers/CsvParser.d.ts +9 -0
- package/dist/parsers/CsvParser.js +110 -0
- package/dist/parsers/ExcelParser.d.ts +2 -2
- package/dist/parsers/ExcelParser.js +145 -114
- package/dist/parsers/HtmlParser.d.ts +2 -0
- package/dist/parsers/HtmlParser.js +539 -0
- package/dist/parsers/MarkdownParser.d.ts +2 -0
- package/dist/parsers/MarkdownParser.js +360 -0
- package/dist/parsers/OpenOfficeParser.d.ts +2 -2
- package/dist/parsers/OpenOfficeParser.js +237 -128
- package/dist/parsers/PdfParser.d.ts +2 -2
- package/dist/parsers/PdfParser.js +52 -49
- package/dist/parsers/PowerPointParser.d.ts +2 -2
- package/dist/parsers/PowerPointParser.js +132 -123
- package/dist/parsers/RtfParser.d.ts +22 -2
- package/dist/parsers/RtfParser.js +1398 -1282
- package/dist/parsers/WordParser.d.ts +3 -2
- package/dist/parsers/WordParser.js +333 -115
- package/dist/sbom.cdx.json +103 -103
- package/dist/types.d.ts +833 -5
- package/dist/types.js +71 -0
- package/dist/utils/astUtils.d.ts +16 -0
- package/dist/utils/astUtils.js +32 -0
- package/dist/utils/configUtils.d.ts +26 -0
- package/dist/utils/configUtils.js +140 -0
- package/dist/utils/envUtils.js +56 -2
- package/dist/utils/errorUtils.d.ts +17 -29
- package/dist/utils/errorUtils.js +109 -52
- package/dist/utils/moduleLoader.js +15 -9
- package/dist/utils/ocrUtils.js +2 -1
- package/dist/utils/sheetUtils.d.ts +7 -0
- package/dist/utils/sheetUtils.js +35 -0
- package/dist/utils/styleMapper.d.ts +36 -0
- package/dist/utils/styleMapper.js +224 -0
- package/dist/utils/xmlUtils.d.ts +0 -8
- package/dist/utils/xmlUtils.js +2 -1
- package/package.json +28 -9
|
@@ -23,6 +23,8 @@
|
|
|
23
23
|
*/
|
|
24
24
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
25
25
|
exports.parseOpenOffice = void 0;
|
|
26
|
+
const types_js_1 = require("../types.js");
|
|
27
|
+
const astUtils_js_1 = require("../utils/astUtils.js");
|
|
26
28
|
const chartUtils_js_1 = require("../utils/chartUtils.js");
|
|
27
29
|
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
28
30
|
const imageUtils_js_1 = require("../utils/imageUtils.js");
|
|
@@ -63,6 +65,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
63
65
|
}
|
|
64
66
|
const mainContentFile = files.find(f => f.path === 'content.xml') || files.find(f => f.path.match(contentFileRegex));
|
|
65
67
|
const stylesFile = files.find(f => f.path === 'styles.xml');
|
|
68
|
+
const stylesDom = stylesFile ? (0, xmlUtils_js_1.parseXmlString)(stylesFile.content.toString()) : undefined;
|
|
66
69
|
const content = [];
|
|
67
70
|
const notes = [];
|
|
68
71
|
// Style Map: styleName -> TextFormatting
|
|
@@ -70,9 +73,13 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
70
73
|
const styleMap = {};
|
|
71
74
|
const paragraphStyleMap = {};
|
|
72
75
|
const listCounters = {}; // Track item index per listId/level
|
|
76
|
+
let currentListId = null;
|
|
77
|
+
let lastListType = null;
|
|
78
|
+
let lastListStyle = null;
|
|
79
|
+
let listIdCounter = 0;
|
|
80
|
+
let lastWasList = false;
|
|
73
81
|
// Helper to parse styles
|
|
74
|
-
const parseStyles = (
|
|
75
|
-
const xml = (0, xmlUtils_js_1.parseXmlString)(xmlString);
|
|
82
|
+
const parseStyles = (xml) => {
|
|
76
83
|
const styles = (0, xmlUtils_js_1.getElementsByTagName)(xml, "style:style");
|
|
77
84
|
for (const style of styles) {
|
|
78
85
|
const name = style.getAttribute("style:name");
|
|
@@ -151,8 +158,8 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
151
158
|
}
|
|
152
159
|
}
|
|
153
160
|
};
|
|
154
|
-
if (
|
|
155
|
-
parseStyles(
|
|
161
|
+
if (stylesDom) {
|
|
162
|
+
parseStyles(stylesDom);
|
|
156
163
|
}
|
|
157
164
|
/**
|
|
158
165
|
* Helper to parse a paragraph node (text:p or text:h) and extract its content.
|
|
@@ -172,9 +179,10 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
172
179
|
*/
|
|
173
180
|
const parseInlineContent = (node, styleMap, config, notes, paragraphStyleMap, parentFormatting = {}, linkMetadata, sourceXml = '') => {
|
|
174
181
|
const children = [];
|
|
182
|
+
const anchorIds = [];
|
|
175
183
|
let fullText = '';
|
|
176
184
|
if (!node.childNodes)
|
|
177
|
-
return { text: '', children: [] };
|
|
185
|
+
return { text: '', children: [], anchorIds: [] };
|
|
178
186
|
for (let i = 0; i < node.childNodes.length; i++) {
|
|
179
187
|
const child = node.childNodes[i];
|
|
180
188
|
if (child.nodeType === 3) { // Text node
|
|
@@ -192,7 +200,12 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
192
200
|
else if ((0, xmlUtils_js_1.isElement)(child)) {
|
|
193
201
|
const element = child;
|
|
194
202
|
const tagName = element.tagName;
|
|
195
|
-
if (tagName === 'text:
|
|
203
|
+
if (tagName === 'text:bookmark' || tagName === 'text:bookmark-start') {
|
|
204
|
+
const name = element.getAttribute('text:name');
|
|
205
|
+
if (name)
|
|
206
|
+
anchorIds.push(name);
|
|
207
|
+
}
|
|
208
|
+
else if (tagName === 'text:s') {
|
|
196
209
|
// Space
|
|
197
210
|
const count = parseInt(element.getAttribute('text:c') || '1');
|
|
198
211
|
const spaces = ' '.repeat(count);
|
|
@@ -221,7 +234,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
221
234
|
type: 'text',
|
|
222
235
|
text: '\n',
|
|
223
236
|
formatting: parentFormatting,
|
|
224
|
-
metadata:
|
|
237
|
+
metadata: { ...(linkMetadata || {}), isLineBreak: true }
|
|
225
238
|
});
|
|
226
239
|
}
|
|
227
240
|
else if (tagName === 'text:span') {
|
|
@@ -231,15 +244,34 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
231
244
|
const spanContent = parseInlineContent(element, styleMap, config, notes, paragraphStyleMap, formatting, linkMetadata, sourceXml);
|
|
232
245
|
fullText += spanContent.text;
|
|
233
246
|
children.push(...spanContent.children);
|
|
247
|
+
anchorIds.push(...spanContent.anchorIds);
|
|
234
248
|
}
|
|
235
249
|
else if (tagName === 'text:a') {
|
|
236
250
|
// Hyperlink
|
|
237
|
-
|
|
238
|
-
const
|
|
239
|
-
const
|
|
251
|
+
let href = element.getAttribute('xlink:href') || '';
|
|
252
|
+
const isInternal = href.startsWith('#');
|
|
253
|
+
const linkType = isInternal ? 'internal' : 'external';
|
|
254
|
+
if (isInternal) {
|
|
255
|
+
// ODT internal links can be encoded and might have suffixes like |outline
|
|
256
|
+
try {
|
|
257
|
+
href = decodeURIComponent(href).split('|')[0];
|
|
258
|
+
}
|
|
259
|
+
catch (e) {
|
|
260
|
+
href = href.split('|')[0];
|
|
261
|
+
}
|
|
262
|
+
// Normalize internal link: if it contains #, keep only from # onwards
|
|
263
|
+
if (href.includes('#')) {
|
|
264
|
+
href = '#' + href.split('#').pop();
|
|
265
|
+
}
|
|
266
|
+
}
|
|
267
|
+
let newLinkMetadata;
|
|
268
|
+
if (!isInternal || !config.ignoreInternalLinks) {
|
|
269
|
+
newLinkMetadata = { link: href, linkType: linkType };
|
|
270
|
+
}
|
|
240
271
|
const linkContent = parseInlineContent(element, styleMap, config, notes, paragraphStyleMap, parentFormatting, newLinkMetadata, sourceXml);
|
|
241
272
|
fullText += linkContent.text;
|
|
242
273
|
children.push(...linkContent.children);
|
|
274
|
+
anchorIds.push(...linkContent.anchorIds);
|
|
243
275
|
}
|
|
244
276
|
else if (tagName === 'text:note' && !config.ignoreNotes) {
|
|
245
277
|
// Footnote or endnote
|
|
@@ -258,7 +290,10 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
258
290
|
type: 'paragraph',
|
|
259
291
|
text: npContent.text,
|
|
260
292
|
children: npContent.children,
|
|
261
|
-
metadata:
|
|
293
|
+
metadata: {
|
|
294
|
+
...(npContent.alignment ? { alignment: npContent.alignment } : {}),
|
|
295
|
+
...(npContent.anchorIds?.length ? { anchorIds: npContent.anchorIds } : {})
|
|
296
|
+
}
|
|
262
297
|
};
|
|
263
298
|
noteChildren.push(npNode);
|
|
264
299
|
}
|
|
@@ -318,7 +353,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
318
353
|
}
|
|
319
354
|
}
|
|
320
355
|
}
|
|
321
|
-
return { text: fullText, children };
|
|
356
|
+
return { text: fullText, children, anchorIds };
|
|
322
357
|
};
|
|
323
358
|
/**
|
|
324
359
|
* Helper to parse a paragraph node (text:p or text:h) and extract its content.
|
|
@@ -389,7 +424,32 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
389
424
|
}
|
|
390
425
|
}
|
|
391
426
|
}
|
|
392
|
-
return { text: content.text, children: content.children, alignment, style: paraStyle || undefined };
|
|
427
|
+
return { text: content.text, children: content.children, alignment, style: paraStyle || undefined, anchorIds: content.anchorIds };
|
|
428
|
+
};
|
|
429
|
+
/**
|
|
430
|
+
* Splits paragraph content into multiple segments based on line breaks.
|
|
431
|
+
* Used to handle soft line breaks within list items.
|
|
432
|
+
*
|
|
433
|
+
* @param pContent - The content of a single paragraph
|
|
434
|
+
* @returns Array of content segments
|
|
435
|
+
*/
|
|
436
|
+
const splitParagraphByBreaks = (pContent) => {
|
|
437
|
+
const segments = [];
|
|
438
|
+
let currentText = "";
|
|
439
|
+
let currentChildren = [];
|
|
440
|
+
for (const child of pContent.children) {
|
|
441
|
+
if (child.type === "text" && child.metadata?.isLineBreak) {
|
|
442
|
+
segments.push({ text: currentText, children: currentChildren });
|
|
443
|
+
currentText = "";
|
|
444
|
+
currentChildren = [];
|
|
445
|
+
}
|
|
446
|
+
else {
|
|
447
|
+
currentText += child.text || "";
|
|
448
|
+
currentChildren.push(child);
|
|
449
|
+
}
|
|
450
|
+
}
|
|
451
|
+
segments.push({ text: currentText, children: currentChildren });
|
|
452
|
+
return segments;
|
|
393
453
|
};
|
|
394
454
|
/**
|
|
395
455
|
* Helper to parse a table node and extract its structure.
|
|
@@ -603,8 +663,9 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
603
663
|
(0, xmlUtils_js_1.getDirectChildren)(officeBody, "office:spreadsheet")[0];
|
|
604
664
|
if (bodyContent) {
|
|
605
665
|
const bodyChildren = (0, xmlUtils_js_1.getDirectChildren)(bodyContent, "*");
|
|
666
|
+
const isSpreadsheet = bodyContent.tagName === "office:spreadsheet";
|
|
606
667
|
for (const child of bodyChildren) {
|
|
607
|
-
traverse(child, content, false, xmlString);
|
|
668
|
+
traverse(child, content, false, xmlString, isSpreadsheet);
|
|
608
669
|
}
|
|
609
670
|
}
|
|
610
671
|
}
|
|
@@ -616,19 +677,28 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
616
677
|
* @param targetArray - The array to push extracted content nodes to
|
|
617
678
|
* @param forceHeading - If true, treats all paragraphs as headings (used for slide titles)
|
|
618
679
|
* @param sourceXml - The source XML string for raw content extraction
|
|
680
|
+
* @param asSheet - If true, treats tables as sheets (for ODS)
|
|
619
681
|
*/
|
|
620
|
-
function traverse(node, targetArray, forceHeading = false, sourceXml) {
|
|
682
|
+
function traverse(node, targetArray, forceHeading = false, sourceXml, asSheet = false) {
|
|
621
683
|
if (node.tagName === "text:p") {
|
|
622
684
|
const pContent = parseParagraphContent(node, paragraphStyleMap, styleMap, config, sourceXml);
|
|
623
685
|
const type = (forceHeading || (node.getAttribute("text:style-name") || '').toLowerCase().includes('title')) ? 'heading' : 'paragraph';
|
|
686
|
+
const metadata = {
|
|
687
|
+
...(pContent.alignment ? { alignment: pContent.alignment } : {}),
|
|
688
|
+
...(pContent.style ? { style: pContent.style } : {}),
|
|
689
|
+
...(pContent.anchorIds?.length ? { anchorIds: pContent.anchorIds } : {})
|
|
690
|
+
};
|
|
691
|
+
const nodeId = node.getAttribute("xml:id") || node.getAttribute("text:id");
|
|
692
|
+
if (nodeId) {
|
|
693
|
+
if (!metadata.anchorIds)
|
|
694
|
+
metadata.anchorIds = [];
|
|
695
|
+
metadata.anchorIds.push(nodeId);
|
|
696
|
+
}
|
|
624
697
|
const pNode = {
|
|
625
698
|
type,
|
|
626
699
|
text: pContent.text,
|
|
627
700
|
children: pContent.children,
|
|
628
|
-
metadata
|
|
629
|
-
...(pContent.alignment ? { alignment: pContent.alignment } : {}),
|
|
630
|
-
...(pContent.style ? { style: pContent.style } : {})
|
|
631
|
-
}
|
|
701
|
+
metadata
|
|
632
702
|
};
|
|
633
703
|
if (type === 'heading' && pNode.metadata) {
|
|
634
704
|
pNode.metadata.level = pNode.metadata.level || 1;
|
|
@@ -640,42 +710,65 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
640
710
|
pNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
|
|
641
711
|
}
|
|
642
712
|
targetArray.push(pNode);
|
|
713
|
+
lastWasList = false;
|
|
643
714
|
}
|
|
644
715
|
else if (node.tagName === "text:h") {
|
|
645
716
|
const level = parseInt(node.getAttribute("text:outline-level") || "1");
|
|
646
717
|
const hContent = parseParagraphContent(node, paragraphStyleMap, styleMap, config, sourceXml);
|
|
718
|
+
const metadata = {
|
|
719
|
+
level,
|
|
720
|
+
...(hContent.alignment ? { alignment: hContent.alignment } : {}),
|
|
721
|
+
...(hContent.style ? { style: hContent.style } : {}),
|
|
722
|
+
...(hContent.anchorIds?.length ? { anchorIds: hContent.anchorIds } : {})
|
|
723
|
+
};
|
|
724
|
+
const nodeId = node.getAttribute("xml:id") || node.getAttribute("text:id");
|
|
725
|
+
if (nodeId) {
|
|
726
|
+
if (!metadata.anchorIds)
|
|
727
|
+
metadata.anchorIds = [];
|
|
728
|
+
metadata.anchorIds.push(nodeId);
|
|
729
|
+
}
|
|
647
730
|
const hNode = {
|
|
648
731
|
type: 'heading',
|
|
649
732
|
text: hContent.text,
|
|
650
733
|
children: hContent.children,
|
|
651
|
-
metadata
|
|
652
|
-
level,
|
|
653
|
-
...(hContent.alignment ? { alignment: hContent.alignment } : {}),
|
|
654
|
-
...(hContent.style ? { style: hContent.style } : {})
|
|
655
|
-
}
|
|
734
|
+
metadata
|
|
656
735
|
};
|
|
657
736
|
if (config.includeRawContent) {
|
|
658
737
|
hNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
|
|
659
738
|
}
|
|
660
739
|
targetArray.push(hNode);
|
|
740
|
+
lastWasList = false;
|
|
661
741
|
}
|
|
662
742
|
else if (node.tagName === "table:table") {
|
|
663
743
|
// Parse table with proper structure
|
|
664
744
|
const tableNode = parseTable(node, paragraphStyleMap, styleMap, config, sourceXml);
|
|
745
|
+
if (asSheet) {
|
|
746
|
+
tableNode.type = 'sheet';
|
|
747
|
+
const sheetName = node.getAttribute("table:name");
|
|
748
|
+
if (sheetName) {
|
|
749
|
+
tableNode.metadata = { ...tableNode.metadata, sheetName };
|
|
750
|
+
}
|
|
751
|
+
}
|
|
752
|
+
const tableId = node.getAttribute("xml:id") || node.getAttribute("table:name");
|
|
753
|
+
if (tableId) {
|
|
754
|
+
if (!tableNode.metadata)
|
|
755
|
+
tableNode.metadata = {};
|
|
756
|
+
tableNode.metadata.anchorIds = tableNode.metadata.anchorIds || [];
|
|
757
|
+
tableNode.metadata.anchorIds.push(tableId);
|
|
758
|
+
}
|
|
665
759
|
if (config.includeRawContent) {
|
|
666
760
|
tableNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
|
|
667
761
|
}
|
|
668
762
|
targetArray.push(tableNode);
|
|
763
|
+
lastWasList = false;
|
|
669
764
|
}
|
|
670
765
|
else if (node.tagName === "text:list") {
|
|
671
766
|
// Parse list structure with proper listId tracking
|
|
672
767
|
const listItems = (0, xmlUtils_js_1.getDirectChildren)(node, "text:list-item");
|
|
673
|
-
// Get list style name to use as listId (or generate one)
|
|
674
|
-
const listStyleName = node.getAttribute("text:style-name") || node.getAttribute("xml:id");
|
|
675
|
-
const listId = listStyleName || `list-${targetArray.length}`;
|
|
676
768
|
// Determine list type by checking the list style definition
|
|
677
769
|
let listType = 'unordered';
|
|
678
770
|
let isVisible = false;
|
|
771
|
+
const listStyleName = node.getAttribute("text:style-name") || node.getAttribute("xml:id");
|
|
679
772
|
let styleNameToCheck = listStyleName;
|
|
680
773
|
// If no style name, check parent list for inherited style
|
|
681
774
|
if (!styleNameToCheck) {
|
|
@@ -689,9 +782,8 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
689
782
|
parentNode = parentNode.parentNode;
|
|
690
783
|
}
|
|
691
784
|
}
|
|
692
|
-
// Try to find list style in automatic styles to determine type and visibility
|
|
785
|
+
// Try to find list style in automatic styles or styles.xml to determine type and visibility
|
|
693
786
|
if (styleNameToCheck) {
|
|
694
|
-
const automaticStyles = (0, xmlUtils_js_1.getFirstElementByTagName)((0, xmlUtils_js_1.parseXmlString)(mainContentFile?.content.toString() || ''), "office:automatic-styles");
|
|
695
787
|
if (automaticStyles) {
|
|
696
788
|
const listStyles = (0, xmlUtils_js_1.getElementsByTagName)(automaticStyles, "text:list-style");
|
|
697
789
|
for (const listStyle of listStyles) {
|
|
@@ -708,32 +800,32 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
708
800
|
listType = 'unordered';
|
|
709
801
|
isVisible = bulletLevels.some(l => !!l.getAttribute("text:bullet-char"));
|
|
710
802
|
}
|
|
711
|
-
if (imageLevels.length > 0)
|
|
803
|
+
else if (imageLevels.length > 0) {
|
|
804
|
+
listType = 'unordered';
|
|
712
805
|
isVisible = true;
|
|
806
|
+
}
|
|
713
807
|
break;
|
|
714
808
|
}
|
|
715
809
|
}
|
|
716
810
|
}
|
|
717
|
-
|
|
718
|
-
|
|
719
|
-
|
|
720
|
-
|
|
721
|
-
|
|
722
|
-
|
|
723
|
-
|
|
724
|
-
|
|
725
|
-
|
|
726
|
-
|
|
727
|
-
|
|
728
|
-
|
|
729
|
-
|
|
730
|
-
|
|
731
|
-
|
|
732
|
-
|
|
811
|
+
if (!isVisible && stylesDom) {
|
|
812
|
+
const officeStyles = (0, xmlUtils_js_1.getFirstElementByTagName)(stylesDom, "office:styles");
|
|
813
|
+
if (officeStyles) {
|
|
814
|
+
const listStyles = (0, xmlUtils_js_1.getElementsByTagName)(officeStyles, "text:list-style");
|
|
815
|
+
for (const listStyle of listStyles) {
|
|
816
|
+
if (listStyle.getAttribute("style:name") === styleNameToCheck) {
|
|
817
|
+
const bulletLevels = (0, xmlUtils_js_1.getElementsByTagName)(listStyle, "text:list-level-style-bullet");
|
|
818
|
+
const numberLevels = (0, xmlUtils_js_1.getElementsByTagName)(listStyle, "text:list-level-style-number");
|
|
819
|
+
if (numberLevels.length > 0) {
|
|
820
|
+
listType = 'ordered';
|
|
821
|
+
isVisible = numberLevels.some(l => !!l.getAttribute("style:num-format"));
|
|
822
|
+
}
|
|
823
|
+
else if (bulletLevels.length > 0) {
|
|
824
|
+
listType = 'unordered';
|
|
825
|
+
isVisible = bulletLevels.some(l => !!l.getAttribute("text:bullet-char"));
|
|
826
|
+
}
|
|
827
|
+
break;
|
|
733
828
|
}
|
|
734
|
-
if (imageLevels.length > 0)
|
|
735
|
-
isVisible = true;
|
|
736
|
-
break;
|
|
737
829
|
}
|
|
738
830
|
}
|
|
739
831
|
}
|
|
@@ -741,6 +833,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
741
833
|
// If the list is not visible, it's likely a layout list used by Impress.
|
|
742
834
|
// We should traverse its items and treat their content as regular nodes.
|
|
743
835
|
if (!isVisible) {
|
|
836
|
+
lastWasList = false;
|
|
744
837
|
for (let i = 0; i < listItems.length; i++) {
|
|
745
838
|
const item = listItems[i];
|
|
746
839
|
if (item.childNodes) {
|
|
@@ -754,6 +847,24 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
754
847
|
}
|
|
755
848
|
return;
|
|
756
849
|
}
|
|
850
|
+
// List Continuity Logic:
|
|
851
|
+
// If this list follows another list of the same type and style, or we are in ODP and it's sequential,
|
|
852
|
+
// we should reuse the previous listId to maintain numbering.
|
|
853
|
+
const isODP = fileType === 'odp';
|
|
854
|
+
const sameStyle = styleNameToCheck && styleNameToCheck === lastListStyle;
|
|
855
|
+
const sameType = listType === lastListType;
|
|
856
|
+
let listId;
|
|
857
|
+
if (lastWasList && (sameStyle || (isODP && sameType))) {
|
|
858
|
+
listId = currentListId;
|
|
859
|
+
}
|
|
860
|
+
else {
|
|
861
|
+
// New list
|
|
862
|
+
listId = styleNameToCheck || `list-${++listIdCounter}`;
|
|
863
|
+
currentListId = listId;
|
|
864
|
+
lastListType = listType;
|
|
865
|
+
lastListStyle = styleNameToCheck;
|
|
866
|
+
}
|
|
867
|
+
lastWasList = true;
|
|
757
868
|
// Calculate indentation level by counting parent text:list elements
|
|
758
869
|
let indentation = 0;
|
|
759
870
|
let parent = node.parentNode;
|
|
@@ -774,59 +885,57 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
774
885
|
// Process each list item
|
|
775
886
|
for (let i = 0; i < listItems.length; i++) {
|
|
776
887
|
const item = listItems[i];
|
|
777
|
-
|
|
778
|
-
listCounters[listId][indentKey]++;
|
|
779
|
-
const itemIndex = listCounters[listId][indentKey];
|
|
780
|
-
// Reset deeper levels when we encounter an item at this level
|
|
781
|
-
for (let k = indentation + 1; k < 10; k++) {
|
|
782
|
-
if (listCounters[listId][k.toString()] !== undefined) {
|
|
783
|
-
listCounters[listId][k.toString()] = -1;
|
|
784
|
-
}
|
|
785
|
-
}
|
|
888
|
+
let hasIndexedThisItem = false;
|
|
786
889
|
// Iterate over direct children of list item (paragraphs, headings, nested lists)
|
|
787
890
|
if (item.childNodes) {
|
|
788
891
|
for (let j = 0; j < item.childNodes.length; j++) {
|
|
789
892
|
const child = item.childNodes[j];
|
|
790
893
|
if ((0, xmlUtils_js_1.isElement)(child)) { // Element
|
|
791
894
|
const element = child;
|
|
792
|
-
if (element.tagName === "text:p") {
|
|
793
|
-
|
|
794
|
-
|
|
795
|
-
|
|
796
|
-
|
|
797
|
-
|
|
798
|
-
|
|
799
|
-
|
|
800
|
-
indentation,
|
|
801
|
-
itemIndex,
|
|
802
|
-
listId,
|
|
803
|
-
alignment: pContent.alignment || 'left',
|
|
804
|
-
style: pContent.style
|
|
895
|
+
if (element.tagName === "text:p" || element.tagName === "text:h") {
|
|
896
|
+
if (!hasIndexedThisItem) {
|
|
897
|
+
listCounters[listId][indentKey]++;
|
|
898
|
+
hasIndexedThisItem = true;
|
|
899
|
+
for (let k = indentation + 1; k < 10; k++) {
|
|
900
|
+
if (listCounters[listId][k.toString()] !== undefined) {
|
|
901
|
+
listCounters[listId][k.toString()] = -1;
|
|
902
|
+
}
|
|
805
903
|
}
|
|
806
|
-
}
|
|
807
|
-
|
|
808
|
-
|
|
809
|
-
|
|
810
|
-
|
|
811
|
-
|
|
812
|
-
|
|
813
|
-
|
|
814
|
-
|
|
815
|
-
|
|
816
|
-
|
|
817
|
-
|
|
818
|
-
|
|
819
|
-
|
|
820
|
-
|
|
821
|
-
|
|
822
|
-
|
|
823
|
-
|
|
824
|
-
|
|
904
|
+
}
|
|
905
|
+
const itemIndex = listCounters[listId][indentKey];
|
|
906
|
+
const pContent = parseParagraphContent(element, paragraphStyleMap, styleMap, config, sourceXml);
|
|
907
|
+
const segments = splitParagraphByBreaks(pContent);
|
|
908
|
+
for (let k = 0; k < segments.length; k++) {
|
|
909
|
+
const segment = segments[k];
|
|
910
|
+
if (!segment.text.trim() && segment.children.length === 0)
|
|
911
|
+
continue;
|
|
912
|
+
const isFirst = k === 0;
|
|
913
|
+
const nodeType = isFirst ? 'list' : 'paragraph';
|
|
914
|
+
const node = {
|
|
915
|
+
type: nodeType,
|
|
916
|
+
text: segment.text,
|
|
917
|
+
children: segment.children,
|
|
918
|
+
metadata: isFirst ? {
|
|
919
|
+
listType,
|
|
920
|
+
indentation,
|
|
921
|
+
itemIndex,
|
|
922
|
+
listId,
|
|
923
|
+
alignment: pContent.alignment || 'left',
|
|
924
|
+
style: pContent.style
|
|
925
|
+
} : {
|
|
926
|
+
alignment: pContent.alignment || 'left',
|
|
927
|
+
style: pContent.style
|
|
928
|
+
}
|
|
929
|
+
};
|
|
930
|
+
// Special case for headings in lists
|
|
931
|
+
if (isFirst && element.tagName === "text:h") {
|
|
932
|
+
const level = parseInt(element.getAttribute("text:outline-level") || "1");
|
|
933
|
+
node.metadata.level = level;
|
|
825
934
|
}
|
|
826
|
-
|
|
827
|
-
|
|
828
|
-
|
|
829
|
-
|
|
935
|
+
if (config.includeRawContent)
|
|
936
|
+
node.rawContent = (0, xmlUtils_js_1.getRawContent)(element, sourceXml, config);
|
|
937
|
+
targetArray.push(node);
|
|
938
|
+
}
|
|
830
939
|
}
|
|
831
940
|
else if (element.tagName === "text:list") {
|
|
832
941
|
// Recursive call for nested list
|
|
@@ -871,14 +980,19 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
871
980
|
const parts = imageHref.split('/');
|
|
872
981
|
imageHref = parts[parts.length - 1];
|
|
873
982
|
}
|
|
983
|
+
const metadata = {
|
|
984
|
+
attachmentName: imageHref,
|
|
985
|
+
...(altText ? { altText } : {})
|
|
986
|
+
};
|
|
987
|
+
const frameId = node.getAttribute("xml:id") || node.getAttribute("draw:name");
|
|
988
|
+
if (frameId) {
|
|
989
|
+
metadata.anchorIds = [frameId];
|
|
990
|
+
}
|
|
874
991
|
const imageNode = {
|
|
875
992
|
type: 'image',
|
|
876
993
|
text: '',
|
|
877
994
|
children: [],
|
|
878
|
-
metadata
|
|
879
|
-
attachmentName: imageHref,
|
|
880
|
-
...(altText ? { altText } : {})
|
|
881
|
-
}
|
|
995
|
+
metadata
|
|
882
996
|
};
|
|
883
997
|
if (config.includeRawContent) {
|
|
884
998
|
imageNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
|
|
@@ -1211,10 +1325,10 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1211
1325
|
if (config.ocr) {
|
|
1212
1326
|
if (attachment.mimeType.startsWith('image/')) {
|
|
1213
1327
|
try {
|
|
1214
|
-
attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(media.content, {
|
|
1328
|
+
attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(media.content, { ...config.ocrConfig })).trim();
|
|
1215
1329
|
}
|
|
1216
1330
|
catch (e) {
|
|
1217
|
-
(0, errorUtils_js_1.logWarning)(
|
|
1331
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.OCR_FAILED, config, attachment.name, e);
|
|
1218
1332
|
}
|
|
1219
1333
|
}
|
|
1220
1334
|
}
|
|
@@ -1351,7 +1465,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1351
1465
|
node.text = attachment.ocrText;
|
|
1352
1466
|
}
|
|
1353
1467
|
if (attachment.chartData && node.type === 'chart') {
|
|
1354
|
-
node.text = attachment.chartData.rawTexts.join(config.newlineDelimiter
|
|
1468
|
+
node.text = attachment.chartData.rawTexts.join(config.newlineDelimiter);
|
|
1355
1469
|
}
|
|
1356
1470
|
}
|
|
1357
1471
|
}
|
|
@@ -1385,32 +1499,27 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1385
1499
|
if (config.putNotesAtLast && notes.length > 0) {
|
|
1386
1500
|
content.push(...notes);
|
|
1387
1501
|
}
|
|
1388
|
-
|
|
1389
|
-
|
|
1390
|
-
|
|
1391
|
-
|
|
1392
|
-
|
|
1393
|
-
|
|
1394
|
-
|
|
1395
|
-
|
|
1396
|
-
|
|
1397
|
-
|
|
1398
|
-
|
|
1399
|
-
|
|
1400
|
-
|
|
1401
|
-
|
|
1402
|
-
|
|
1403
|
-
|
|
1404
|
-
|
|
1405
|
-
|
|
1406
|
-
|
|
1407
|
-
|
|
1408
|
-
|
|
1409
|
-
|
|
1410
|
-
return t;
|
|
1411
|
-
};
|
|
1412
|
-
return getText(c);
|
|
1413
|
-
}).filter(t => t != '').join(config.newlineDelimiter ?? '\n')
|
|
1414
|
-
};
|
|
1502
|
+
const toTextSync = () => content.map(c => {
|
|
1503
|
+
const getText = (node) => {
|
|
1504
|
+
let t = '';
|
|
1505
|
+
if (node.children && node.children.length > 0) {
|
|
1506
|
+
// Check if children have their own children (container vs leaf)
|
|
1507
|
+
// If children are leaf nodes (text/image), join with empty string
|
|
1508
|
+
// If children are container nodes (paragraphs/rows), join with newline
|
|
1509
|
+
const hasGrandChildren = node.children.some(child => child.children && child.children.length > 0);
|
|
1510
|
+
const separator = hasGrandChildren ? config.newlineDelimiter : '';
|
|
1511
|
+
t += node.children.map(getText).filter(t => t != '').join(separator);
|
|
1512
|
+
}
|
|
1513
|
+
else {
|
|
1514
|
+
t += node.text || '';
|
|
1515
|
+
}
|
|
1516
|
+
return t;
|
|
1517
|
+
};
|
|
1518
|
+
return getText(c);
|
|
1519
|
+
}).filter(t => t != '').join(config.newlineDelimiter);
|
|
1520
|
+
return (0, astUtils_js_1.createAST)(fileType, {
|
|
1521
|
+
...metadata,
|
|
1522
|
+
styleMap: combinedStyleMap
|
|
1523
|
+
}, content, attachments, config, toTextSync);
|
|
1415
1524
|
};
|
|
1416
1525
|
exports.parseOpenOffice = parseOpenOffice;
|
|
@@ -56,7 +56,7 @@
|
|
|
56
56
|
* @see https://mozilla.github.io/pdf.js/ PDF.js documentation
|
|
57
57
|
* @see https://www.adobe.com/devnet/pdf/pdf_reference.html PDF Reference
|
|
58
58
|
*/
|
|
59
|
-
import {
|
|
59
|
+
import { FullOfficeParserConfig, OfficeParserAST } from '../types.js';
|
|
60
60
|
/**
|
|
61
61
|
* Parses a PDF file and extracts content.
|
|
62
62
|
*
|
|
@@ -64,4 +64,4 @@ import { OfficeParserAST, OfficeParserConfig } from '../types.js';
|
|
|
64
64
|
* @param config - Parser configuration
|
|
65
65
|
* @returns Promise resolving to the parsed AST
|
|
66
66
|
*/
|
|
67
|
-
export declare const parsePdf: (buffer: Buffer, config:
|
|
67
|
+
export declare const parsePdf: (buffer: Buffer, config: FullOfficeParserConfig) => Promise<OfficeParserAST>;
|