officeparser 5.2.2 → 6.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1399 @@
1
+ "use strict";
2
+ /**
3
+ * OpenDocument Format (ODF) Parser
4
+ *
5
+ * **ODF Overview:**
6
+ * ODF is an open standard for office documents (ISO/IEC 26300).
7
+ * Used by LibreOffice, OpenOffice, and other applications.
8
+ *
9
+ * **File Structure:**
10
+ * ODF files are ZIP archives containing:
11
+ * - `mimetype` - File type identification
12
+ * - `content.xml` - Main document content
13
+ * - `styles.xml` - Style definitions
14
+ * - `meta.xml` - Document metadata
15
+ * - `Pictures/*` - Embedded images
16
+ *
17
+ * **Supported Formats:**
18
+ * - ODT: Text documents (application/vnd.oasis.opendocument.text)
19
+ * - ODP: Presentations (application/vnd.oasis.opendocument.presentation)
20
+ * - ODS: Spreadsheets (application/vnd.oasis.opendocument.spreadsheet)
21
+ *
22
+ * @module OpenOfficeParser
23
+ */
24
+ Object.defineProperty(exports, "__esModule", { value: true });
25
+ exports.parseOpenOffice = void 0;
26
+ const chartUtils_1 = require("../utils/chartUtils");
27
+ const errorUtils_1 = require("../utils/errorUtils");
28
+ const imageUtils_1 = require("../utils/imageUtils");
29
+ const ocrUtils_1 = require("../utils/ocrUtils");
30
+ const xmlUtils_1 = require("../utils/xmlUtils");
31
+ const zipUtils_1 = require("../utils/zipUtils");
32
+ /**
33
+ * Parses an OpenOffice document (.odt, .odp, .ods) and extracts content.
34
+ *
35
+ * @param buffer - The ODF file as a Buffer
36
+ * @param config - Parser configuration
37
+ * @returns A promise resolving to the parsed AST
38
+ */
39
+ const parseOpenOffice = async (buffer, config) => {
40
+ const contentFileRegex = /content\.xml/;
41
+ const objectContentFileRegex = /Object \d+\/content\.xml/;
42
+ const mediaFileRegex = /(Pictures|media)\/.*/;
43
+ const metaFileRegex = /meta\.xml/;
44
+ const stylesFileRegex = /styles\.xml/;
45
+ const mimetypeFileRegex = /mimetype/;
46
+ const files = await (0, zipUtils_1.extractFiles)(buffer, x => !!x.match(contentFileRegex) ||
47
+ !!x.match(objectContentFileRegex) ||
48
+ !!x.match(metaFileRegex) ||
49
+ !!x.match(stylesFileRegex) ||
50
+ !!x.match(mimetypeFileRegex) ||
51
+ (!!config.extractAttachments && !!x.match(mediaFileRegex)));
52
+ // 1. Determine File Type
53
+ const mimetypeFile = files.find(f => f.path === 'mimetype');
54
+ let fileType = 'odt'; // Default
55
+ if (mimetypeFile) {
56
+ const mime = mimetypeFile.content.toString().trim();
57
+ if (mime.includes('spreadsheet'))
58
+ fileType = 'ods';
59
+ else if (mime.includes('presentation'))
60
+ fileType = 'odp';
61
+ else if (mime.includes('text'))
62
+ fileType = 'odt';
63
+ }
64
+ const mainContentFile = files.find(f => f.path === 'content.xml') || files.find(f => f.path.match(contentFileRegex));
65
+ const stylesFile = files.find(f => f.path === 'styles.xml');
66
+ const content = [];
67
+ const notes = [];
68
+ // Style Map: styleName -> TextFormatting
69
+ // Inline style parsing (from content.xml automatic styles)
70
+ const styleMap = {};
71
+ const paragraphStyleMap = {};
72
+ const listCounters = {}; // Track item index per listId/level
73
+ // Helper to parse styles
74
+ const parseStyles = (xmlString) => {
75
+ const xml = (0, xmlUtils_1.parseXmlString)(xmlString);
76
+ const styles = (0, xmlUtils_1.getElementsByTagName)(xml, "style:style");
77
+ for (const style of styles) {
78
+ const name = style.getAttribute("style:name");
79
+ if (!name)
80
+ continue;
81
+ const styleInfo = {};
82
+ // Parse paragraph properties for alignment and drop caps
83
+ const paraProps = (0, xmlUtils_1.getElementsByTagName)(style, "style:paragraph-properties")[0];
84
+ if (paraProps) {
85
+ const textAlign = paraProps.getAttribute("fo:text-align");
86
+ if (textAlign) {
87
+ const alignMap = {
88
+ 'start': 'left',
89
+ 'left': 'left',
90
+ 'center': 'center',
91
+ 'end': 'right',
92
+ 'right': 'right',
93
+ 'justify': 'justify'
94
+ };
95
+ if (alignMap[textAlign]) {
96
+ styleInfo.alignment = alignMap[textAlign];
97
+ }
98
+ }
99
+ // Detect Drop Caps
100
+ const dropCap = (0, xmlUtils_1.getElementsByTagName)(paraProps, "style:drop-cap")[0];
101
+ if (dropCap) {
102
+ styleInfo.dropCap = true;
103
+ }
104
+ }
105
+ if (Object.keys(styleInfo).length > 0) {
106
+ paragraphStyleMap[name] = styleInfo;
107
+ }
108
+ // Parse text properties
109
+ const textProps = (0, xmlUtils_1.getElementsByTagName)(style, "style:text-properties")[0];
110
+ // Parse table cell properties (for ODS background)
111
+ const cellProps = (0, xmlUtils_1.getElementsByTagName)(style, "style:table-cell-properties")[0];
112
+ const formatting = {};
113
+ if (cellProps) {
114
+ const bgColor = cellProps.getAttribute("fo:background-color");
115
+ if (bgColor && bgColor !== 'transparent')
116
+ formatting.backgroundColor = bgColor;
117
+ }
118
+ if (textProps) {
119
+ if (textProps.getAttribute("fo:font-weight") === "bold" || textProps.getAttribute("style:font-weight-asian") === "bold")
120
+ formatting.bold = true;
121
+ if (textProps.getAttribute("fo:font-style") === "italic" || textProps.getAttribute("style:font-style-asian") === "italic")
122
+ formatting.italic = true;
123
+ if (textProps.getAttribute("style:text-underline-style") === "solid")
124
+ formatting.underline = true;
125
+ if (textProps.getAttribute("style:text-line-through-style") === "solid")
126
+ formatting.strikethrough = true;
127
+ const size = textProps.getAttribute("fo:font-size") || textProps.getAttribute("style:font-size-asian");
128
+ if (size)
129
+ formatting.size = size;
130
+ const color = textProps.getAttribute("fo:color");
131
+ if (color)
132
+ formatting.color = color;
133
+ // Background color (text level) - override cell level if present?
134
+ const bgColor = textProps.getAttribute("fo:background-color");
135
+ if (bgColor && bgColor !== 'transparent')
136
+ formatting.backgroundColor = bgColor;
137
+ // Font family
138
+ const fontName = textProps.getAttribute("style:font-name") || textProps.getAttribute("fo:font-family");
139
+ if (fontName)
140
+ formatting.font = fontName;
141
+ // Subscript/Superscript from text-position (e.g., "sub 58%" or "super 58%")
142
+ const textPosition = textProps.getAttribute("style:text-position");
143
+ if (textPosition) {
144
+ if (textPosition.startsWith("sub"))
145
+ formatting.subscript = true;
146
+ if (textPosition.startsWith("super"))
147
+ formatting.superscript = true;
148
+ }
149
+ if (Object.keys(formatting).length > 0)
150
+ styleMap[name] = formatting;
151
+ }
152
+ }
153
+ };
154
+ if (stylesFile) {
155
+ parseStyles(stylesFile.content.toString());
156
+ }
157
+ /**
158
+ * Helper to parse a paragraph node (text:p or text:h) and extract its content.
159
+ * Returns the paragraph content without creating a content node.
160
+ *
161
+ * @param node - The paragraph element to parse
162
+ * Helper to parse inline content (text, spans, links, notes, etc.) recursively.
163
+ *
164
+ * @param node - The element to parse (paragraph, span, or link)
165
+ * @param styleMap - Map of style names to formatting
166
+ * @param config - Parser configuration
167
+ * @param notes - Optional array to collect footnotes/endnotes
168
+ * @param paragraphStyleMap - Map of style names to alignments and props (needed for notes)
169
+ * @param parentFormatting - Formatting inherited from parent (e.g. span inside span)
170
+ * @param linkMetadata - Metadata inherited from parent link
171
+ * @returns Object containing text and children
172
+ */
173
+ const parseInlineContent = (node, styleMap, config, notes, paragraphStyleMap, parentFormatting = {}, linkMetadata) => {
174
+ const children = [];
175
+ let fullText = '';
176
+ if (!node.childNodes)
177
+ return { text: '', children: [] };
178
+ for (let i = 0; i < node.childNodes.length; i++) {
179
+ const child = node.childNodes[i];
180
+ if (child.nodeType === 3) { // Text node
181
+ const text = child.textContent || '';
182
+ if (text) {
183
+ fullText += text;
184
+ children.push({
185
+ type: 'text',
186
+ text: text,
187
+ formatting: parentFormatting,
188
+ metadata: linkMetadata ? { ...linkMetadata } : undefined
189
+ });
190
+ }
191
+ }
192
+ else if (child.nodeType === 1) {
193
+ const element = child;
194
+ const tagName = element.tagName;
195
+ if (tagName === 'text:s') {
196
+ // Space
197
+ const count = parseInt(element.getAttribute('text:c') || '1');
198
+ const spaces = ' '.repeat(count);
199
+ fullText += spaces;
200
+ children.push({
201
+ type: 'text',
202
+ text: spaces,
203
+ formatting: parentFormatting,
204
+ metadata: linkMetadata ? { ...linkMetadata } : undefined
205
+ });
206
+ }
207
+ else if (tagName === 'text:tab') {
208
+ // Tab
209
+ fullText += '\t';
210
+ children.push({
211
+ type: 'text',
212
+ text: '\t',
213
+ formatting: parentFormatting,
214
+ metadata: linkMetadata ? { ...linkMetadata } : undefined
215
+ });
216
+ }
217
+ else if (tagName === 'text:line-break') {
218
+ // Line break
219
+ fullText += '\n';
220
+ children.push({
221
+ type: 'text',
222
+ text: '\n',
223
+ formatting: parentFormatting,
224
+ metadata: linkMetadata ? { ...linkMetadata } : undefined
225
+ });
226
+ }
227
+ else if (tagName === 'text:span') {
228
+ // Formatted text span
229
+ const styleName = element.getAttribute("text:style-name");
230
+ const formatting = styleName ? { ...parentFormatting, ...styleMap[styleName] } : parentFormatting;
231
+ const spanContent = parseInlineContent(element, styleMap, config, notes, paragraphStyleMap, formatting, linkMetadata);
232
+ fullText += spanContent.text;
233
+ children.push(...spanContent.children);
234
+ }
235
+ else if (tagName === 'text:a') {
236
+ // Hyperlink
237
+ const href = element.getAttribute('xlink:href') || '';
238
+ const linkType = href.startsWith('#') ? 'internal' : 'external';
239
+ const newLinkMetadata = { link: href, linkType: linkType };
240
+ const linkContent = parseInlineContent(element, styleMap, config, notes, paragraphStyleMap, parentFormatting, newLinkMetadata);
241
+ fullText += linkContent.text;
242
+ children.push(...linkContent.children);
243
+ }
244
+ else if (tagName === 'text:note' && !config.ignoreNotes) {
245
+ // Footnote or endnote
246
+ const noteClass = (element.getAttribute('text:note-class') || 'footnote');
247
+ const noteId = element.getAttribute('text:id') || element.getAttribute('xml:id') || undefined;
248
+ const noteBody = (0, xmlUtils_1.getElementsByTagName)(element, "text:note-body")[0];
249
+ if (noteBody) {
250
+ // Extract note content recursively
251
+ const notePs = (0, xmlUtils_1.getElementsByTagName)(noteBody, "text:p");
252
+ const noteChildren = [];
253
+ let noteText = '';
254
+ for (const np of notePs) {
255
+ const npContent = parseParagraphContent(np, paragraphStyleMap, styleMap, config);
256
+ noteText += (noteText ? ' ' : '') + npContent.text;
257
+ const npNode = {
258
+ type: 'paragraph',
259
+ text: npContent.text,
260
+ children: npContent.children,
261
+ metadata: npContent.alignment ? { alignment: npContent.alignment } : undefined
262
+ };
263
+ noteChildren.push(npNode);
264
+ }
265
+ const noteNode = {
266
+ type: 'note',
267
+ text: noteText,
268
+ children: noteChildren,
269
+ metadata: {
270
+ noteType: noteClass,
271
+ noteId: noteId
272
+ }
273
+ };
274
+ if (config.putNotesAtLast) {
275
+ notes.push(noteNode);
276
+ }
277
+ else {
278
+ children.push(noteNode);
279
+ }
280
+ }
281
+ }
282
+ else if (tagName === 'draw:frame') {
283
+ // Inline image
284
+ const frame = element;
285
+ // Extract alt text
286
+ let altText = '';
287
+ const svgTitle = (0, xmlUtils_1.getElementsByTagName)(frame, "svg:title")[0];
288
+ const svgDesc = (0, xmlUtils_1.getElementsByTagName)(frame, "svg:desc")[0];
289
+ if (svgTitle && svgTitle.textContent) {
290
+ altText = svgTitle.textContent;
291
+ }
292
+ else if (svgDesc && svgDesc.textContent) {
293
+ altText = svgDesc.textContent;
294
+ }
295
+ // Extract image href
296
+ let imageHref = '';
297
+ const drawImages = (0, xmlUtils_1.getElementsByTagName)(frame, "draw:image");
298
+ if (drawImages.length > 0) {
299
+ imageHref = drawImages[0].getAttribute("xlink:href") || '';
300
+ if (imageHref) {
301
+ const parts = imageHref.split('/');
302
+ imageHref = parts[parts.length - 1];
303
+ }
304
+ }
305
+ const imageNode = {
306
+ type: 'image',
307
+ text: '',
308
+ children: [],
309
+ metadata: {
310
+ attachmentName: imageHref,
311
+ ...(altText ? { altText } : {})
312
+ }
313
+ };
314
+ if (config.includeRawContent) {
315
+ imageNode.rawContent = frame.toString();
316
+ }
317
+ children.push(imageNode);
318
+ }
319
+ }
320
+ }
321
+ return { text: fullText, children };
322
+ };
323
+ /**
324
+ * Helper to parse a paragraph node (text:p or text:h) and extract its content.
325
+ * Returns the paragraph content without creating a content node.
326
+ *
327
+ * @param node - The paragraph element to parse
328
+ * @param paraStyleMap - Map of style names to alignments/props
329
+ * @param styleMap - Map of style names to formatting
330
+ * @param config - Parser configuration
331
+ * @returns Object containing text, children, alignment, and style info
332
+ */
333
+ const parseParagraphContent = (node, paraStyleMap, styleMap, config) => {
334
+ // Get paragraph style for alignment and drop caps
335
+ const paraStyle = node.getAttribute("text:style-name");
336
+ const styleInfo = paraStyle ? paraStyleMap[paraStyle] : undefined;
337
+ const alignment = styleInfo?.alignment;
338
+ const dropCap = styleInfo?.dropCap;
339
+ // Parse content recursively using the new helper
340
+ const content = parseInlineContent(node, styleMap, config, notes, paraStyleMap);
341
+ // Add style name to metadata of children if they don't have one
342
+ if (paraStyle) {
343
+ content.children.forEach(child => {
344
+ if (child.type === 'text') {
345
+ if (!child.metadata)
346
+ child.metadata = {};
347
+ // Only add style if it's a text node and doesn't have one?
348
+ // Or just add it.
349
+ // Cast to any to avoid union type issues for now, or check type
350
+ const meta = child.metadata;
351
+ if (!meta.style)
352
+ meta.style = paraStyle;
353
+ }
354
+ });
355
+ }
356
+ // Fallback: if no children were created but there's text content
357
+ if (content.children.length === 0 && node.textContent) {
358
+ const fullText = node.textContent;
359
+ if (fullText.trim()) {
360
+ content.text = fullText;
361
+ content.children.push({
362
+ type: 'text',
363
+ text: fullText
364
+ });
365
+ }
366
+ }
367
+ // Handle Drop Cap: Apply large font to first letter if configured
368
+ if (dropCap && content.children.length > 0) {
369
+ const firstChild = content.children[0];
370
+ if (firstChild.type === 'text' && firstChild.text) {
371
+ if (firstChild.text.length === 1) {
372
+ // Already a single letter, just apply formatting
373
+ firstChild.formatting = { ...firstChild.formatting, size: '58.5pt' };
374
+ }
375
+ else {
376
+ // Split text node
377
+ const firstChar = firstChild.text[0];
378
+ const restText = firstChild.text.substring(1);
379
+ const dropCapNode = {
380
+ type: 'text',
381
+ text: firstChar,
382
+ formatting: { ...firstChild.formatting, size: '58.5pt' },
383
+ metadata: firstChild.metadata
384
+ };
385
+ // Update original node
386
+ firstChild.text = restText;
387
+ // Insert drop cap node
388
+ content.children.unshift(dropCapNode);
389
+ }
390
+ }
391
+ }
392
+ return { text: content.text, children: content.children, alignment, style: paraStyle || undefined };
393
+ };
394
+ /**
395
+ * Helper to parse a table node and extract its structure.
396
+ * Properly creates table → row → cell hierarchy with metadata.
397
+ *
398
+ * @param tableNode - The table:table element
399
+ * @param paraStyleMap - Map of style names to alignments
400
+ * @param styleMap - Map of style names to formatting
401
+ * @param config - Parser configuration
402
+ * @returns Table content node with proper structure
403
+ */
404
+ const parseTable = (tableNode, paraStyleMap, styleMap, config) => {
405
+ const rows = [];
406
+ // Use getDirectChildren to avoid nested table rows
407
+ const tableRows = (0, xmlUtils_1.getDirectChildren)(tableNode, "table:table-row");
408
+ let rowIndex = 0;
409
+ for (const row of tableRows) {
410
+ const cells = [];
411
+ // Use getDirectChildren to avoid nested table cells
412
+ const tableCells = (0, xmlUtils_1.getDirectChildren)(row, "table:table-cell");
413
+ const rowsRepeated = parseInt(row.getAttribute("table:number-rows-repeated") || "1");
414
+ let colIndex = 0;
415
+ for (const cell of tableCells) {
416
+ const cellChildren = [];
417
+ let cellTextRef = { value: '' };
418
+ const colsRepeated = parseInt(cell.getAttribute("table:number-columns-repeated") || "1");
419
+ const colSpan = parseInt(cell.getAttribute("table:number-columns-spanned") || "1");
420
+ const rowSpan = parseInt(cell.getAttribute("table:number-rows-spanned") || "1");
421
+ // Helper to recursively process cell children (handles frames, text-boxes, etc. in ODP)
422
+ const processChildren = (node) => {
423
+ if (!node.childNodes)
424
+ return;
425
+ for (let i = 0; i < node.childNodes.length; i++) {
426
+ const child = node.childNodes[i];
427
+ if (child.nodeType === 1) { // Element
428
+ const element = child;
429
+ if (element.tagName === "text:p" || element.tagName === "text:h") {
430
+ const pContent = parseParagraphContent(element, paraStyleMap, styleMap, config);
431
+ const pNode = {
432
+ type: element.tagName === "text:h" ? 'heading' : 'paragraph',
433
+ text: pContent.text,
434
+ children: pContent.children,
435
+ metadata: {
436
+ ...(pContent.alignment ? { alignment: pContent.alignment } : {}),
437
+ ...(pContent.style ? { style: pContent.style } : {})
438
+ }
439
+ };
440
+ // Clean up metadata if empty
441
+ if (Object.keys(pNode.metadata || {}).length === 0)
442
+ delete pNode.metadata;
443
+ if (element.tagName === "text:h") {
444
+ if (!pNode.metadata)
445
+ pNode.metadata = {};
446
+ pNode.metadata.level = parseInt(element.getAttribute("text:outline-level") || "1");
447
+ }
448
+ if (config.includeRawContent) {
449
+ pNode.rawContent = element.toString();
450
+ }
451
+ cellChildren.push(pNode);
452
+ cellTextRef.value += pContent.text;
453
+ // Add newline if there are multiple paragraphs/headings
454
+ if (cellTextRef.value && !cellTextRef.value.endsWith('\n')) {
455
+ cellTextRef.value += '\n';
456
+ }
457
+ }
458
+ else if (element.tagName === "table:table") {
459
+ // Recursive call for nested table
460
+ const nestedTableNode = parseTable(element, paraStyleMap, styleMap, config);
461
+ cellChildren.push(nestedTableNode);
462
+ }
463
+ else if (element.tagName === "draw:frame" || element.tagName === "draw:text-box") {
464
+ // Recursively process container content (common in ODP)
465
+ processChildren(element);
466
+ }
467
+ }
468
+ }
469
+ };
470
+ processChildren(cell);
471
+ let cellText = cellTextRef.value;
472
+ // Trim trailing newline from cellText
473
+ if (cellText.endsWith('\n')) {
474
+ cellText = cellText.slice(0, -1);
475
+ }
476
+ // Add cell(s) for repeated columns
477
+ for (let k = 0; k < colsRepeated; k++) {
478
+ const cellNode = {
479
+ type: 'cell',
480
+ text: cellText,
481
+ children: cellChildren.length > 0 ? (k === 0 ? cellChildren : JSON.parse(JSON.stringify(cellChildren))) : [],
482
+ metadata: { row: rowIndex, col: colIndex }
483
+ };
484
+ const cellMetadata = cellNode.metadata;
485
+ if (colSpan > 1)
486
+ cellMetadata.colSpan = colSpan;
487
+ if (rowSpan > 1)
488
+ cellMetadata.rowSpan = rowSpan;
489
+ if (config.includeRawContent) {
490
+ cellNode.rawContent = cell.toString();
491
+ }
492
+ cells.push(cellNode);
493
+ colIndex++;
494
+ }
495
+ }
496
+ // Add row(s) for repeated rows
497
+ for (let k = 0; k < rowsRepeated; k++) {
498
+ const rowNode = {
499
+ type: 'row',
500
+ children: k === 0 ? cells : JSON.parse(JSON.stringify(cells))
501
+ };
502
+ // Fix row indices for repeated rows
503
+ if (k > 0) {
504
+ rowNode.children?.forEach(c => {
505
+ if (c.metadata && 'row' in c.metadata) {
506
+ c.metadata.row = rowIndex;
507
+ }
508
+ });
509
+ }
510
+ if (config.includeRawContent) {
511
+ rowNode.rawContent = row.toString();
512
+ }
513
+ rows.push(rowNode);
514
+ rowIndex++;
515
+ }
516
+ }
517
+ return {
518
+ type: 'table',
519
+ children: rows
520
+ };
521
+ };
522
+ const parseContentXml = (xmlString) => {
523
+ const xml = (0, xmlUtils_1.parseXmlString)(xmlString);
524
+ const body = (0, xmlUtils_1.getElementsByTagName)(xml, "office:body")[0];
525
+ // Parse automatic styles (local to content.xml)
526
+ const automaticStyles = (0, xmlUtils_1.getElementsByTagName)(xml, "office:automatic-styles")[0];
527
+ if (automaticStyles) {
528
+ const styles = (0, xmlUtils_1.getElementsByTagName)(automaticStyles, "style:style");
529
+ for (const style of styles) {
530
+ const name = style.getAttribute("style:name");
531
+ if (!name)
532
+ continue;
533
+ // Parse paragraph properties for alignment
534
+ const paraProps = (0, xmlUtils_1.getElementsByTagName)(style, "style:paragraph-properties")[0];
535
+ const styleInfo = {};
536
+ if (paraProps) {
537
+ const textAlign = paraProps.getAttribute("fo:text-align");
538
+ if (textAlign) {
539
+ const alignMap = {
540
+ 'start': 'left',
541
+ 'left': 'left',
542
+ 'center': 'center',
543
+ 'end': 'right',
544
+ 'right': 'right',
545
+ 'justify': 'justify'
546
+ };
547
+ if (alignMap[textAlign]) {
548
+ styleInfo.alignment = alignMap[textAlign];
549
+ }
550
+ }
551
+ const dropCap = (0, xmlUtils_1.getElementsByTagName)(paraProps, "style:drop-cap")[0];
552
+ if (dropCap)
553
+ styleInfo.dropCap = true;
554
+ }
555
+ if (Object.keys(styleInfo).length > 0) {
556
+ paragraphStyleMap[name] = styleInfo;
557
+ }
558
+ const textProps = (0, xmlUtils_1.getElementsByTagName)(style, "style:text-properties")[0];
559
+ if (textProps) {
560
+ const formatting = {};
561
+ if (textProps.getAttribute("fo:font-weight") === "bold" || textProps.getAttribute("style:font-weight-asian") === "bold")
562
+ formatting.bold = true;
563
+ if (textProps.getAttribute("fo:font-style") === "italic" || textProps.getAttribute("style:font-style-asian") === "italic")
564
+ formatting.italic = true;
565
+ if (textProps.getAttribute("style:text-underline-style") === "solid")
566
+ formatting.underline = true;
567
+ if (textProps.getAttribute("style:text-line-through-style") === "solid")
568
+ formatting.strikethrough = true;
569
+ const size = textProps.getAttribute("fo:font-size") || textProps.getAttribute("style:font-size-asian");
570
+ if (size)
571
+ formatting.size = size;
572
+ const color = textProps.getAttribute("fo:color");
573
+ if (color)
574
+ formatting.color = color;
575
+ // Background color
576
+ const bgColor = textProps.getAttribute("fo:background-color");
577
+ if (bgColor && bgColor !== 'transparent')
578
+ formatting.backgroundColor = bgColor;
579
+ // Font family
580
+ const fontName = textProps.getAttribute("style:font-name") || textProps.getAttribute("fo:font-family");
581
+ if (fontName)
582
+ formatting.font = fontName;
583
+ // Subscript/Superscript from text-position (e.g., "sub 58%" or "super 58%")
584
+ const textPosition = textProps.getAttribute("style:text-position");
585
+ if (textPosition) {
586
+ if (textPosition.startsWith("sub"))
587
+ formatting.subscript = true;
588
+ if (textPosition.startsWith("super"))
589
+ formatting.superscript = true;
590
+ }
591
+ if (Object.keys(formatting).length > 0)
592
+ styleMap[name] = formatting;
593
+ }
594
+ }
595
+ }
596
+ /**
597
+ * Recursively traverses a node and its children to extract content.
598
+ * Properly handles paragraphs, headings, tables, lists, and frames.
599
+ *
600
+ * @param node - The element to traverse
601
+ * @param targetArray - The array to push extracted content nodes to
602
+ * @param forceHeading - If true, treats all paragraphs as headings (used for slide titles)
603
+ */
604
+ const traverse = (node, targetArray, forceHeading = false) => {
605
+ if (node.tagName === "text:p") {
606
+ const pContent = parseParagraphContent(node, paragraphStyleMap, styleMap, config);
607
+ const type = (forceHeading || (node.getAttribute("text:style-name") || '').toLowerCase().includes('title')) ? 'heading' : 'paragraph';
608
+ const pNode = {
609
+ type,
610
+ text: pContent.text,
611
+ children: pContent.children,
612
+ metadata: {
613
+ ...(pContent.alignment ? { alignment: pContent.alignment } : {}),
614
+ ...(pContent.style ? { style: pContent.style } : {})
615
+ }
616
+ };
617
+ if (type === 'heading' && pNode.metadata) {
618
+ pNode.metadata.level = pNode.metadata.level || 1;
619
+ }
620
+ // Clean up metadata if empty
621
+ if (Object.keys(pNode.metadata || {}).length === 0)
622
+ delete pNode.metadata;
623
+ if (config.includeRawContent) {
624
+ pNode.rawContent = node.toString();
625
+ }
626
+ targetArray.push(pNode);
627
+ }
628
+ else if (node.tagName === "text:h") {
629
+ const level = parseInt(node.getAttribute("text:outline-level") || "1");
630
+ const hContent = parseParagraphContent(node, paragraphStyleMap, styleMap, config);
631
+ const hNode = {
632
+ type: 'heading',
633
+ text: hContent.text,
634
+ children: hContent.children,
635
+ metadata: {
636
+ level,
637
+ ...(hContent.alignment ? { alignment: hContent.alignment } : {}),
638
+ ...(hContent.style ? { style: hContent.style } : {})
639
+ }
640
+ };
641
+ if (config.includeRawContent) {
642
+ hNode.rawContent = node.toString();
643
+ }
644
+ targetArray.push(hNode);
645
+ }
646
+ else if (node.tagName === "table:table") {
647
+ // Parse table with proper structure
648
+ const tableNode = parseTable(node, paragraphStyleMap, styleMap, config);
649
+ if (config.includeRawContent) {
650
+ tableNode.rawContent = node.toString();
651
+ }
652
+ targetArray.push(tableNode);
653
+ }
654
+ else if (node.tagName === "text:list") {
655
+ // Parse list structure with proper listId tracking
656
+ const listItems = (0, xmlUtils_1.getDirectChildren)(node, "text:list-item");
657
+ // Get list style name to use as listId (or generate one)
658
+ const listStyleName = node.getAttribute("text:style-name") || node.getAttribute("xml:id");
659
+ const listId = listStyleName || `list-${targetArray.length}`;
660
+ // Determine list type by checking the list style definition
661
+ let listType = 'unordered';
662
+ let isVisible = false;
663
+ let styleNameToCheck = listStyleName;
664
+ // If no style name, check parent list for inherited style
665
+ if (!styleNameToCheck) {
666
+ let parentNode = node.parentNode;
667
+ while (parentNode && !styleNameToCheck) {
668
+ if (parentNode.nodeName === 'text:list') {
669
+ styleNameToCheck = parentNode.getAttribute("text:style-name");
670
+ if (styleNameToCheck)
671
+ break;
672
+ }
673
+ parentNode = parentNode.parentNode;
674
+ }
675
+ }
676
+ // Try to find list style in automatic styles to determine type and visibility
677
+ if (styleNameToCheck) {
678
+ const automaticStyles = (0, xmlUtils_1.getElementsByTagName)((0, xmlUtils_1.parseXmlString)(mainContentFile?.content.toString() || ''), "office:automatic-styles")[0];
679
+ if (automaticStyles) {
680
+ const listStyles = (0, xmlUtils_1.getElementsByTagName)(automaticStyles, "text:list-style");
681
+ for (const listStyle of listStyles) {
682
+ if (listStyle.getAttribute("style:name") === styleNameToCheck) {
683
+ // Check if it has bullet or number level styles
684
+ const bulletLevels = (0, xmlUtils_1.getElementsByTagName)(listStyle, "text:list-level-style-bullet");
685
+ const numberLevels = (0, xmlUtils_1.getElementsByTagName)(listStyle, "text:list-level-style-number");
686
+ const imageLevels = (0, xmlUtils_1.getElementsByTagName)(listStyle, "text:list-level-style-image");
687
+ if (numberLevels.length > 0) {
688
+ listType = 'ordered';
689
+ isVisible = numberLevels.some(l => !!l.getAttribute("style:num-format"));
690
+ }
691
+ else if (bulletLevels.length > 0) {
692
+ listType = 'unordered';
693
+ isVisible = bulletLevels.some(l => !!l.getAttribute("text:bullet-char"));
694
+ }
695
+ if (imageLevels.length > 0)
696
+ isVisible = true;
697
+ break;
698
+ }
699
+ }
700
+ }
701
+ // Also check in styles.xml if still unordered and hidden
702
+ if (stylesFile && !isVisible) {
703
+ const stylesXml = (0, xmlUtils_1.parseXmlString)(stylesFile.content.toString());
704
+ const listStyles = (0, xmlUtils_1.getElementsByTagName)(stylesXml, "text:list-style");
705
+ for (const listStyle of listStyles) {
706
+ if (listStyle.getAttribute("style:name") === styleNameToCheck) {
707
+ const bulletLevels = (0, xmlUtils_1.getElementsByTagName)(listStyle, "text:list-level-style-bullet");
708
+ const numberLevels = (0, xmlUtils_1.getElementsByTagName)(listStyle, "text:list-level-style-number");
709
+ const imageLevels = (0, xmlUtils_1.getElementsByTagName)(listStyle, "text:list-level-style-image");
710
+ if (numberLevels.length > 0) {
711
+ listType = 'ordered';
712
+ isVisible = numberLevels.some(l => !!l.getAttribute("style:num-format"));
713
+ }
714
+ else if (bulletLevels.length > 0) {
715
+ listType = 'unordered';
716
+ isVisible = bulletLevels.some(l => !!l.getAttribute("text:bullet-char"));
717
+ }
718
+ if (imageLevels.length > 0)
719
+ isVisible = true;
720
+ break;
721
+ }
722
+ }
723
+ }
724
+ }
725
+ // If the list is not visible, it's likely a layout list used by Impress.
726
+ // We should traverse its items and treat their content as regular nodes.
727
+ if (!isVisible) {
728
+ for (let i = 0; i < listItems.length; i++) {
729
+ const item = listItems[i];
730
+ if (item.childNodes) {
731
+ for (let j = 0; j < item.childNodes.length; j++) {
732
+ const child = item.childNodes[j];
733
+ if (child.nodeType === 1) { // Element
734
+ traverse(child, targetArray, forceHeading);
735
+ }
736
+ }
737
+ }
738
+ }
739
+ return;
740
+ }
741
+ // Calculate indentation level by counting parent text:list elements
742
+ let indentation = 0;
743
+ let parent = node.parentNode;
744
+ while (parent) {
745
+ if (parent.nodeName === 'text:list') {
746
+ indentation++;
747
+ }
748
+ parent = parent.parentNode;
749
+ }
750
+ // Track list counters for this listId (similar to WordParser)
751
+ if (!listCounters[listId]) {
752
+ listCounters[listId] = {};
753
+ }
754
+ const indentKey = indentation.toString();
755
+ if (listCounters[listId][indentKey] === undefined) {
756
+ listCounters[listId][indentKey] = -1; // Will increment to 0 on first item
757
+ }
758
+ // Process each list item
759
+ for (let i = 0; i < listItems.length; i++) {
760
+ const item = listItems[i];
761
+ // Increment item index for this list/level
762
+ listCounters[listId][indentKey]++;
763
+ const itemIndex = listCounters[listId][indentKey];
764
+ // Reset deeper levels when we encounter an item at this level
765
+ for (let k = indentation + 1; k < 10; k++) {
766
+ if (listCounters[listId][k.toString()] !== undefined) {
767
+ listCounters[listId][k.toString()] = -1;
768
+ }
769
+ }
770
+ // Iterate over direct children of list item (paragraphs, headings, nested lists)
771
+ if (item.childNodes) {
772
+ for (let j = 0; j < item.childNodes.length; j++) {
773
+ const child = item.childNodes[j];
774
+ if (child.nodeType === 1) { // Element
775
+ const element = child;
776
+ if (element.tagName === "text:p") {
777
+ const pContent = parseParagraphContent(element, paragraphStyleMap, styleMap, config);
778
+ const listNode = {
779
+ type: 'list',
780
+ text: pContent.text,
781
+ children: pContent.children,
782
+ metadata: {
783
+ listType,
784
+ indentation,
785
+ itemIndex,
786
+ listId,
787
+ alignment: pContent.alignment || 'left',
788
+ style: pContent.style
789
+ }
790
+ };
791
+ if (config.includeRawContent)
792
+ listNode.rawContent = element.toString();
793
+ targetArray.push(listNode);
794
+ }
795
+ else if (element.tagName === "text:h") {
796
+ const level = parseInt(element.getAttribute("text:outline-level") || "1");
797
+ const hContent = parseParagraphContent(element, paragraphStyleMap, styleMap, config);
798
+ const listNode = {
799
+ type: 'list',
800
+ text: hContent.text,
801
+ children: hContent.children,
802
+ metadata: {
803
+ listType,
804
+ indentation,
805
+ itemIndex,
806
+ listId,
807
+ ...(hContent.alignment ? { alignment: hContent.alignment } : {}),
808
+ style: hContent.style
809
+ }
810
+ };
811
+ if (config.includeRawContent)
812
+ listNode.rawContent = element.toString();
813
+ targetArray.push(listNode);
814
+ }
815
+ else if (element.tagName === "text:list") {
816
+ // Recursive call for nested list
817
+ traverse(element, targetArray, forceHeading);
818
+ }
819
+ }
820
+ }
821
+ }
822
+ }
823
+ }
824
+ else if (node.tagName === "draw:frame") {
825
+ const presClass = node.getAttribute("presentation:class");
826
+ const isHeading = presClass === "title" || presClass === "sub-title";
827
+ // In presentations, frames often contain text-boxes, images, tables, or objects
828
+ const textBox = (0, xmlUtils_1.getElementsByTagName)(node, "draw:text-box")[0];
829
+ const image = (0, xmlUtils_1.getElementsByTagName)(node, "draw:image")[0];
830
+ const table = (0, xmlUtils_1.getElementsByTagName)(node, "table:table")[0];
831
+ const object = (0, xmlUtils_1.getElementsByTagName)(node, "draw:object")[0];
832
+ if (textBox) {
833
+ traverse(textBox, targetArray, isHeading || forceHeading);
834
+ }
835
+ else if (table) {
836
+ const tableNode = parseTable(table, paragraphStyleMap, styleMap, config);
837
+ if (config.includeRawContent)
838
+ tableNode.rawContent = table.toString();
839
+ targetArray.push(tableNode);
840
+ }
841
+ else if (image) {
842
+ // Extract alt text from svg:title or svg:desc
843
+ let altText = '';
844
+ const svgTitle = (0, xmlUtils_1.getElementsByTagName)(node, "svg:title")[0];
845
+ const svgDesc = (0, xmlUtils_1.getElementsByTagName)(node, "svg:desc")[0];
846
+ if (svgTitle && svgTitle.textContent) {
847
+ altText = svgTitle.textContent;
848
+ }
849
+ else if (svgDesc && svgDesc.textContent) {
850
+ altText = svgDesc.textContent;
851
+ }
852
+ // Extract image href to link to attachment
853
+ let imageHref = image.getAttribute("xlink:href") || '';
854
+ if (imageHref) {
855
+ const parts = imageHref.split('/');
856
+ imageHref = parts[parts.length - 1];
857
+ }
858
+ const imageNode = {
859
+ type: 'image',
860
+ text: '',
861
+ children: [],
862
+ metadata: {
863
+ attachmentName: imageHref,
864
+ ...(altText ? { altText } : {})
865
+ }
866
+ };
867
+ if (config.includeRawContent) {
868
+ imageNode.rawContent = node.toString();
869
+ }
870
+ targetArray.push(imageNode);
871
+ }
872
+ else if (object) {
873
+ // Handle embedded objects like charts
874
+ const href = object.getAttribute("xlink:href");
875
+ if (href) {
876
+ const attachmentName = href.split('/')[0];
877
+ const objectPath = `${attachmentName}/content.xml`;
878
+ const objectFile = files.find(f => f.path === objectPath || f.path.endsWith(objectPath));
879
+ if (objectFile) {
880
+ const chartData = (0, chartUtils_1.extractChartData)(objectFile.content);
881
+ const chartNode = {
882
+ type: 'chart',
883
+ text: chartData.rawTexts.join(" "),
884
+ metadata: {
885
+ attachmentName: attachmentName,
886
+ chartData
887
+ }
888
+ };
889
+ if (config.includeRawContent)
890
+ chartNode.rawContent = node.toString();
891
+ targetArray.push(chartNode);
892
+ }
893
+ else {
894
+ const chartNode = {
895
+ type: 'chart',
896
+ text: "",
897
+ metadata: { attachmentName: attachmentName }
898
+ };
899
+ if (config.includeRawContent)
900
+ chartNode.rawContent = node.toString();
901
+ targetArray.push(chartNode);
902
+ }
903
+ }
904
+ }
905
+ }
906
+ else {
907
+ if (node.childNodes) {
908
+ for (let i = 0; i < node.childNodes.length; i++) {
909
+ const child = node.childNodes[i];
910
+ if (child.nodeType === 1) { // Element
911
+ traverse(child, targetArray, forceHeading);
912
+ }
913
+ }
914
+ }
915
+ }
916
+ };
917
+ // ODS: Spreadsheet
918
+ if (fileType === 'ods') {
919
+ const spreadsheet = (0, xmlUtils_1.getElementsByTagName)(body, "office:spreadsheet")[0];
920
+ if (spreadsheet) {
921
+ const tables = (0, xmlUtils_1.getElementsByTagName)(spreadsheet, "table:table");
922
+ for (let i = 0; i < tables.length; i++) {
923
+ const table = tables[i];
924
+ const sheetName = table.getAttribute("table:name") || `Sheet${i + 1}`;
925
+ const rows = [];
926
+ const tableRows = (0, xmlUtils_1.getElementsByTagName)(table, "table:table-row");
927
+ let rowIndex = 0;
928
+ for (let r = 0; r < tableRows.length; r++) {
929
+ const row = tableRows[r];
930
+ const cells = [];
931
+ const tableCells = (0, xmlUtils_1.getElementsByTagName)(row, "table:table-cell");
932
+ let colIndex = 0;
933
+ const rowsRepeated = parseInt(row.getAttribute("table:number-rows-repeated") || "1");
934
+ for (let c = 0; c < tableCells.length; c++) {
935
+ const cell = tableCells[c];
936
+ const colsRepeated = parseInt(cell.getAttribute("table:number-columns-repeated") || "1");
937
+ // Extract text from cell (paragraphs inside cell)
938
+ let cellText = "";
939
+ const children = [];
940
+ const ps = (0, xmlUtils_1.getElementsByTagName)(cell, "text:p");
941
+ for (let p = 0; p < ps.length; p++) {
942
+ const para = ps[p];
943
+ // Parse text:span elements for formatted text
944
+ const spans = (0, xmlUtils_1.getElementsByTagName)(para, "text:span");
945
+ if (spans.length > 0) {
946
+ for (const span of spans) {
947
+ const styleName = span.getAttribute("text:style-name");
948
+ const formatting = styleName ? styleMap[styleName] : {};
949
+ const text = span.textContent || '';
950
+ cellText += text;
951
+ const textNode = {
952
+ type: 'text',
953
+ text: text,
954
+ formatting: formatting
955
+ };
956
+ children.push(textNode);
957
+ }
958
+ }
959
+ else {
960
+ // No spans - just direct text content
961
+ const text = para.textContent || '';
962
+ cellText += text;
963
+ if (text.trim()) {
964
+ const textNode = {
965
+ type: 'text',
966
+ text: text,
967
+ formatting: {}
968
+ };
969
+ children.push(textNode);
970
+ }
971
+ }
972
+ if (p < ps.length - 1)
973
+ cellText += "\n";
974
+ }
975
+ // Check for embedded draw:frame (images) in cell
976
+ const drawFrames = (0, xmlUtils_1.getElementsByTagName)(cell, "draw:frame");
977
+ for (const frame of drawFrames) {
978
+ // Extract alt text from svg:title or svg:desc
979
+ let altText = '';
980
+ const svgTitle = (0, xmlUtils_1.getElementsByTagName)(frame, "svg:title")[0];
981
+ const svgDesc = (0, xmlUtils_1.getElementsByTagName)(frame, "svg:desc")[0];
982
+ if (svgTitle && svgTitle.textContent) {
983
+ altText = svgTitle.textContent;
984
+ }
985
+ else if (svgDesc && svgDesc.textContent) {
986
+ altText = svgDesc.textContent;
987
+ }
988
+ // Extract image href
989
+ let imageHref = '';
990
+ const drawImages = (0, xmlUtils_1.getElementsByTagName)(frame, "draw:image");
991
+ if (drawImages.length > 0) {
992
+ const rawHref = drawImages[0].getAttribute("xlink:href");
993
+ if (rawHref) {
994
+ const parts = rawHref.split('/');
995
+ imageHref = parts[parts.length - 1];
996
+ }
997
+ }
998
+ // Extract chart object href
999
+ let chartHref = '';
1000
+ const drawObjects = (0, xmlUtils_1.getElementsByTagName)(frame, "draw:object");
1001
+ if (drawObjects.length > 0) {
1002
+ const href = drawObjects[0].getAttribute("xlink:href");
1003
+ if (href) {
1004
+ // Object href is usually "./Object 1"
1005
+ chartHref = href.split('/')[0];
1006
+ }
1007
+ }
1008
+ if (drawImages.length > 0) {
1009
+ // logic for image node
1010
+ const imageNode = {
1011
+ type: 'image',
1012
+ text: '',
1013
+ children: [],
1014
+ metadata: {
1015
+ attachmentName: imageHref,
1016
+ ...(altText ? { altText } : {})
1017
+ }
1018
+ };
1019
+ if (config.includeRawContent) {
1020
+ imageNode.rawContent = frame.toString();
1021
+ }
1022
+ children.push(imageNode);
1023
+ }
1024
+ else if (chartHref) {
1025
+ const chartNode = {
1026
+ type: 'chart',
1027
+ text: '',
1028
+ children: [],
1029
+ metadata: {
1030
+ attachmentName: chartHref
1031
+ }
1032
+ };
1033
+ children.push(chartNode);
1034
+ }
1035
+ }
1036
+ // Add cell(s)
1037
+ for (let k = 0; k < colsRepeated; k++) {
1038
+ // For ODS (spreadsheets), we skip empty cells to avoid massive ASTs (millions of cells)
1039
+ // but for ODP/ODT (presentation/text), cells are part of a defined table grid
1040
+ // Also include cells that have children (e.g., image nodes) even if no text
1041
+ if (cellText || children.length > 0 || fileType !== 'ods') {
1042
+ const cellNode = {
1043
+ type: 'cell',
1044
+ text: cellText,
1045
+ children: children,
1046
+ metadata: { row: rowIndex, col: colIndex }
1047
+ };
1048
+ if (config.includeRawContent) {
1049
+ cellNode.rawContent = cell.toString();
1050
+ }
1051
+ cells.push(cellNode);
1052
+ }
1053
+ colIndex++;
1054
+ }
1055
+ }
1056
+ // Add row(s)
1057
+ if (cells.length > 0) {
1058
+ for (let k = 0; k < rowsRepeated; k++) {
1059
+ const rowNode = {
1060
+ type: 'row',
1061
+ children: JSON.parse(JSON.stringify(cells)),
1062
+ metadata: undefined
1063
+ };
1064
+ // Fix row index in metadata for repeated rows
1065
+ if (k > 0) {
1066
+ rowNode.children?.forEach(c => {
1067
+ if (c.metadata && 'row' in c.metadata) {
1068
+ c.metadata.row = rowIndex;
1069
+ }
1070
+ });
1071
+ }
1072
+ if (config.includeRawContent) {
1073
+ rowNode.rawContent = row.toString();
1074
+ }
1075
+ rows.push(rowNode);
1076
+ rowIndex++;
1077
+ }
1078
+ }
1079
+ else {
1080
+ rowIndex += rowsRepeated;
1081
+ }
1082
+ }
1083
+ const sheetNode = {
1084
+ type: 'sheet',
1085
+ children: rows,
1086
+ metadata: { sheetName }
1087
+ };
1088
+ if (config.includeRawContent) {
1089
+ sheetNode.rawContent = table.toString();
1090
+ }
1091
+ content.push(sheetNode);
1092
+ }
1093
+ }
1094
+ }
1095
+ // ODP: Presentation
1096
+ else if (fileType === 'odp') {
1097
+ const presentation = (0, xmlUtils_1.getElementsByTagName)(body, "office:presentation")[0];
1098
+ if (presentation) {
1099
+ const pages = (0, xmlUtils_1.getDirectChildren)(presentation, "draw:page");
1100
+ const odpNotes = [];
1101
+ for (let i = 0; i < pages.length; i++) {
1102
+ const page = pages[i];
1103
+ const slideNode = {
1104
+ type: 'slide',
1105
+ children: [],
1106
+ metadata: { slideNumber: i + 1 }
1107
+ };
1108
+ // Separate page content and notes
1109
+ let noteNode = undefined;
1110
+ const pageChildren = page.childNodes;
1111
+ if (pageChildren) {
1112
+ for (let j = 0; j < pageChildren.length; j++) {
1113
+ const child = pageChildren[j];
1114
+ if (child.nodeType === 1) { // Element
1115
+ const element = child;
1116
+ if (element.tagName === "presentation:notes") {
1117
+ if (!config.ignoreNotes) {
1118
+ noteNode = {
1119
+ type: 'note',
1120
+ children: [],
1121
+ metadata: {
1122
+ slideNumber: i + 1,
1123
+ noteId: `slide-note-${i + 1}`
1124
+ }
1125
+ };
1126
+ traverse(element, noteNode.children);
1127
+ }
1128
+ continue;
1129
+ }
1130
+ traverse(element, slideNode.children);
1131
+ }
1132
+ }
1133
+ }
1134
+ if (config.includeRawContent) {
1135
+ slideNode.rawContent = page.toString();
1136
+ }
1137
+ content.push(slideNode);
1138
+ if (noteNode && noteNode.children && noteNode.children.length > 0) {
1139
+ if (config.putNotesAtLast) {
1140
+ odpNotes.push(noteNode);
1141
+ }
1142
+ else {
1143
+ content.push(noteNode);
1144
+ }
1145
+ }
1146
+ }
1147
+ if (odpNotes.length > 0) {
1148
+ content.push(...odpNotes);
1149
+ }
1150
+ }
1151
+ }
1152
+ // ODT: Text Document (and generic fallback)
1153
+ else {
1154
+ const textDoc = (0, xmlUtils_1.getElementsByTagName)(body, "office:text")[0];
1155
+ if (textDoc) {
1156
+ traverse(textDoc, content);
1157
+ }
1158
+ }
1159
+ };
1160
+ if (mainContentFile) {
1161
+ parseContentXml(mainContentFile.content.toString());
1162
+ }
1163
+ // Attachments
1164
+ const attachments = [];
1165
+ const mediaFiles = files.filter(f => f.path.match(/(Pictures|media)\/.*/));
1166
+ // ODP/ODT Chart Extraction
1167
+ if (config.extractAttachments) {
1168
+ const objectFiles = files.filter(f => f.path.match(/Object \d+\/content\.xml/));
1169
+ for (const objFile of objectFiles) {
1170
+ const objXml = (0, xmlUtils_1.parseXmlString)(objFile.content.toString());
1171
+ const isChart = (0, xmlUtils_1.getElementsByTagName)(objXml, "chart:chart").length > 0;
1172
+ if (isChart) {
1173
+ const objectId = objFile.path.split('/')[0];
1174
+ const attachment = {
1175
+ type: 'chart',
1176
+ mimeType: 'application/vnd.oasis.opendocument.chart',
1177
+ data: objFile.content.toString('base64'),
1178
+ name: objectId,
1179
+ extension: 'xml'
1180
+ };
1181
+ // Extract data from chart XML
1182
+ const chartData = (0, chartUtils_1.extractChartData)(objFile.content);
1183
+ if (chartData.rawTexts.length > 0) {
1184
+ attachment.chartData = chartData;
1185
+ }
1186
+ attachments.push(attachment);
1187
+ }
1188
+ }
1189
+ }
1190
+ if (config.extractAttachments) {
1191
+ for (const media of mediaFiles) {
1192
+ const attachment = (0, imageUtils_1.createAttachment)(media.path.split('/').pop() || 'image', media.content);
1193
+ attachments.push(attachment);
1194
+ if (config.ocr) {
1195
+ if (attachment.mimeType.startsWith('image/')) {
1196
+ try {
1197
+ attachment.ocrText = (await (0, ocrUtils_1.performOcr)(media.content, config.ocrLanguage)).trim();
1198
+ }
1199
+ catch (e) {
1200
+ (0, errorUtils_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
1201
+ }
1202
+ }
1203
+ }
1204
+ }
1205
+ }
1206
+ const metaFile = files.find(f => f.path.match(metaFileRegex));
1207
+ const metadata = metaFile ? (0, xmlUtils_1.parseOfficeMetadata)(metaFile.content.toString()) : {};
1208
+ // Helper: Resolve ODS chart cell references to actual values
1209
+ // ODS charts often link to cell ranges (e.g., [Sheet1.$A$1:.$A$5]) instead of embedding values
1210
+ const resolveChartReferences = (chartData, nodes) => {
1211
+ const getValuesFromReference = (ref) => {
1212
+ // Remove brackets: [Sheet.$A$1:.$A$5] -> Sheet.$A$1:.$A$5
1213
+ const cleanRef = ref.replace(/^\[|\]$/g, '');
1214
+ const [startPart, endPart] = cleanRef.split(':');
1215
+ const lastDotIdx = startPart.lastIndexOf('.');
1216
+ if (lastDotIdx === -1)
1217
+ return [ref];
1218
+ const sheetName = startPart.substring(0, lastDotIdx).replace(/^'|'$/g, '');
1219
+ const startCoord = startPart.substring(lastDotIdx + 1).replace(/\$/g, '');
1220
+ let endCoord = startCoord;
1221
+ if (endPart) {
1222
+ if (endPart.startsWith('.')) {
1223
+ endCoord = endPart.substring(1).replace(/\$/g, '');
1224
+ }
1225
+ else {
1226
+ const endLastDotIdx = endPart.lastIndexOf('.');
1227
+ endCoord = endPart.substring(endLastDotIdx + 1).replace(/\$/g, '');
1228
+ }
1229
+ }
1230
+ const parseCoord = (coord) => {
1231
+ const colMatch = coord.match(/[A-Z]+/);
1232
+ const rowMatch = coord.match(/\d+/);
1233
+ if (!colMatch || !rowMatch)
1234
+ return null;
1235
+ const colStr = colMatch[0];
1236
+ let colIdx = 0;
1237
+ for (let i = 0; i < colStr.length; i++) {
1238
+ colIdx = colIdx * 26 + (colStr.charCodeAt(i) - 'A'.charCodeAt(0) + 1);
1239
+ }
1240
+ colIdx -= 1;
1241
+ const rowIdx = parseInt(rowMatch[0]) - 1;
1242
+ return { r: rowIdx, c: colIdx };
1243
+ };
1244
+ const start = parseCoord(startCoord);
1245
+ const end = parseCoord(endCoord);
1246
+ if (!start || !end)
1247
+ return [ref];
1248
+ const sheet = nodes.find(n => n.type === 'sheet' && n.metadata?.sheetName === sheetName);
1249
+ if (!sheet || !sheet.children)
1250
+ return [ref];
1251
+ const values = [];
1252
+ // Collect all matching cells
1253
+ for (const row of sheet.children) {
1254
+ if (row.children) {
1255
+ for (const cell of row.children) {
1256
+ const meta = cell.metadata;
1257
+ if (meta && meta.row >= start.r && meta.row <= end.r && meta.col >= start.c && meta.col <= end.c) {
1258
+ values.push(cell.text || '');
1259
+ }
1260
+ }
1261
+ }
1262
+ }
1263
+ return values.length > 0 ? values : [];
1264
+ };
1265
+ // Resolve DataSets
1266
+ for (const ds of chartData.dataSets) {
1267
+ const newValues = [];
1268
+ for (const val of ds.values) {
1269
+ if (val.startsWith('['))
1270
+ newValues.push(...getValuesFromReference(val));
1271
+ else
1272
+ newValues.push(val);
1273
+ }
1274
+ ds.values = newValues;
1275
+ }
1276
+ // Resolve Labels
1277
+ const newLabels = [];
1278
+ for (const label of chartData.labels) {
1279
+ if (label.startsWith('['))
1280
+ newLabels.push(...getValuesFromReference(label));
1281
+ else
1282
+ newLabels.push(label);
1283
+ }
1284
+ chartData.labels = newLabels;
1285
+ // Rebuild rawTexts
1286
+ chartData.rawTexts = [];
1287
+ if (chartData.title)
1288
+ chartData.rawTexts.push(chartData.title);
1289
+ for (const ds of chartData.dataSets) {
1290
+ if (ds.name)
1291
+ chartData.rawTexts.push(ds.name);
1292
+ chartData.rawTexts.push(...chartData.labels);
1293
+ chartData.rawTexts.push(...ds.values);
1294
+ }
1295
+ };
1296
+ // Apply resolution to all chart attachments
1297
+ for (const att of attachments) {
1298
+ if (att.type === 'chart' && att.chartData) {
1299
+ resolveChartReferences(att.chartData, content);
1300
+ }
1301
+ }
1302
+ // Link OCR and Chart text to content nodes
1303
+ // Link OCR and Chart text to content nodes (with heuristic for unlinked images)
1304
+ const assignAttachmentData = (nodes) => {
1305
+ // Step 1: Identify unused image attachments globally
1306
+ const usedAttachmentNames = new Set();
1307
+ const traverseForNames = (ns) => {
1308
+ for (const n of ns) {
1309
+ if (n.metadata && 'attachmentName' in n.metadata) {
1310
+ const name = n.metadata.attachmentName;
1311
+ if (name)
1312
+ usedAttachmentNames.add(name);
1313
+ }
1314
+ if (n.children)
1315
+ traverseForNames(n.children);
1316
+ }
1317
+ };
1318
+ traverseForNames(nodes);
1319
+ const unusedImages = attachments.filter(a => a.type === 'image' && a.name && !usedAttachmentNames.has(a.name));
1320
+ let unusedImageIndex = 0;
1321
+ const processNode = (node) => {
1322
+ if ((node.type === 'image' || node.type === 'chart') && node.metadata && 'attachmentName' in node.metadata) {
1323
+ let attachmentName = node.metadata.attachmentName;
1324
+ // Heuristic: If name is empty, try to assign an unused image attachment
1325
+ if (!attachmentName && node.type === 'image' && unusedImageIndex < unusedImages.length) {
1326
+ const fallbackAtt = unusedImages[unusedImageIndex++];
1327
+ attachmentName = fallbackAtt.name;
1328
+ node.metadata.attachmentName = attachmentName;
1329
+ }
1330
+ if (attachmentName) {
1331
+ const attachment = attachments.find(a => a.name === attachmentName);
1332
+ if (attachment) {
1333
+ if (attachment.ocrText) {
1334
+ node.text = attachment.ocrText;
1335
+ }
1336
+ if (attachment.chartData && node.type === 'chart') {
1337
+ node.text = attachment.chartData.rawTexts.join(config.newlineDelimiter ?? '\n');
1338
+ }
1339
+ }
1340
+ }
1341
+ }
1342
+ // Internal recursion
1343
+ if (node.children) {
1344
+ node.children.forEach(processNode);
1345
+ }
1346
+ };
1347
+ nodes.forEach(processNode);
1348
+ };
1349
+ assignAttachmentData(content);
1350
+ // Create combined styleMap for metadata (matches DOCX format)
1351
+ const combinedStyleMap = {};
1352
+ for (const styleName in styleMap) {
1353
+ combinedStyleMap[styleName] = {
1354
+ formatting: styleMap[styleName],
1355
+ alignment: paragraphStyleMap[styleName]?.alignment
1356
+ };
1357
+ }
1358
+ // Also add styles that only have alignment
1359
+ for (const styleName in paragraphStyleMap) {
1360
+ if (!combinedStyleMap[styleName]) {
1361
+ combinedStyleMap[styleName] = {
1362
+ formatting: {},
1363
+ alignment: paragraphStyleMap[styleName]?.alignment
1364
+ };
1365
+ }
1366
+ }
1367
+ // Append notes to content if configured
1368
+ if (config.putNotesAtLast && notes.length > 0) {
1369
+ content.push(...notes);
1370
+ }
1371
+ return {
1372
+ type: fileType,
1373
+ metadata: {
1374
+ ...metadata,
1375
+ styleMap: combinedStyleMap
1376
+ },
1377
+ content: content,
1378
+ attachments: attachments,
1379
+ toText: () => content.map(c => {
1380
+ const getText = (node) => {
1381
+ let t = '';
1382
+ if (node.children && node.children.length > 0) {
1383
+ // Check if children have their own children (container vs leaf)
1384
+ // If children are leaf nodes (text/image), join with empty string
1385
+ // If children are container nodes (paragraphs/rows), join with newline
1386
+ const hasGrandChildren = node.children.some(child => child.children && child.children.length > 0);
1387
+ const separator = hasGrandChildren ? (config.newlineDelimiter ?? '\n') : '';
1388
+ t += node.children.map(getText).filter(t => t != '').join(separator);
1389
+ }
1390
+ else {
1391
+ t += node.text || '';
1392
+ }
1393
+ return t;
1394
+ };
1395
+ return getText(c);
1396
+ }).filter(t => t != '').join(config.newlineDelimiter ?? '\n')
1397
+ };
1398
+ };
1399
+ exports.parseOpenOffice = parseOpenOffice;