officeparser 5.2.2 → 6.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +411 -163
- package/dist/OfficeParser.d.ts +90 -0
- package/dist/OfficeParser.js +217 -0
- package/dist/index.d.ts +51 -0
- package/dist/index.js +108 -0
- package/dist/officeparser.browser.js +165 -0
- package/dist/officeparser.browser.js.map +7 -0
- package/dist/parsers/ExcelParser.d.ts +33 -0
- package/dist/parsers/ExcelParser.js +643 -0
- package/dist/parsers/OpenOfficeParser.d.ts +32 -0
- package/dist/parsers/OpenOfficeParser.js +1399 -0
- package/dist/parsers/PdfParser.d.ts +68 -0
- package/dist/parsers/PdfParser.js +847 -0
- package/dist/parsers/PowerPointParser.d.ts +33 -0
- package/dist/parsers/PowerPointParser.js +778 -0
- package/dist/parsers/RtfParser.d.ts +164 -0
- package/dist/parsers/RtfParser.js +1641 -0
- package/dist/parsers/WordParser.d.ts +79 -0
- package/dist/parsers/WordParser.js +787 -0
- package/dist/types.d.ts +615 -0
- package/dist/types.js +2 -0
- package/dist/utils/chartUtils.d.ts +7 -0
- package/dist/utils/chartUtils.js +255 -0
- package/dist/utils/errorUtils.d.ts +58 -0
- package/dist/utils/errorUtils.js +120 -0
- package/dist/utils/imageUtils.d.ts +67 -0
- package/dist/utils/imageUtils.js +133 -0
- package/dist/utils/ocrUtils.d.ts +39 -0
- package/dist/utils/ocrUtils.js +61 -0
- package/dist/utils/xmlUtils.d.ts +83 -0
- package/dist/utils/xmlUtils.js +158 -0
- package/dist/utils/zipUtils.d.ts +74 -0
- package/dist/utils/zipUtils.js +112 -0
- package/package.json +44 -17
- package/officeParser.js +0 -790
- package/typings/officeParser.d.ts +0 -33
|
@@ -0,0 +1,1399 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* OpenDocument Format (ODF) Parser
|
|
4
|
+
*
|
|
5
|
+
* **ODF Overview:**
|
|
6
|
+
* ODF is an open standard for office documents (ISO/IEC 26300).
|
|
7
|
+
* Used by LibreOffice, OpenOffice, and other applications.
|
|
8
|
+
*
|
|
9
|
+
* **File Structure:**
|
|
10
|
+
* ODF files are ZIP archives containing:
|
|
11
|
+
* - `mimetype` - File type identification
|
|
12
|
+
* - `content.xml` - Main document content
|
|
13
|
+
* - `styles.xml` - Style definitions
|
|
14
|
+
* - `meta.xml` - Document metadata
|
|
15
|
+
* - `Pictures/*` - Embedded images
|
|
16
|
+
*
|
|
17
|
+
* **Supported Formats:**
|
|
18
|
+
* - ODT: Text documents (application/vnd.oasis.opendocument.text)
|
|
19
|
+
* - ODP: Presentations (application/vnd.oasis.opendocument.presentation)
|
|
20
|
+
* - ODS: Spreadsheets (application/vnd.oasis.opendocument.spreadsheet)
|
|
21
|
+
*
|
|
22
|
+
* @module OpenOfficeParser
|
|
23
|
+
*/
|
|
24
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
25
|
+
exports.parseOpenOffice = void 0;
|
|
26
|
+
const chartUtils_1 = require("../utils/chartUtils");
|
|
27
|
+
const errorUtils_1 = require("../utils/errorUtils");
|
|
28
|
+
const imageUtils_1 = require("../utils/imageUtils");
|
|
29
|
+
const ocrUtils_1 = require("../utils/ocrUtils");
|
|
30
|
+
const xmlUtils_1 = require("../utils/xmlUtils");
|
|
31
|
+
const zipUtils_1 = require("../utils/zipUtils");
|
|
32
|
+
/**
|
|
33
|
+
* Parses an OpenOffice document (.odt, .odp, .ods) and extracts content.
|
|
34
|
+
*
|
|
35
|
+
* @param buffer - The ODF file as a Buffer
|
|
36
|
+
* @param config - Parser configuration
|
|
37
|
+
* @returns A promise resolving to the parsed AST
|
|
38
|
+
*/
|
|
39
|
+
const parseOpenOffice = async (buffer, config) => {
|
|
40
|
+
const contentFileRegex = /content\.xml/;
|
|
41
|
+
const objectContentFileRegex = /Object \d+\/content\.xml/;
|
|
42
|
+
const mediaFileRegex = /(Pictures|media)\/.*/;
|
|
43
|
+
const metaFileRegex = /meta\.xml/;
|
|
44
|
+
const stylesFileRegex = /styles\.xml/;
|
|
45
|
+
const mimetypeFileRegex = /mimetype/;
|
|
46
|
+
const files = await (0, zipUtils_1.extractFiles)(buffer, x => !!x.match(contentFileRegex) ||
|
|
47
|
+
!!x.match(objectContentFileRegex) ||
|
|
48
|
+
!!x.match(metaFileRegex) ||
|
|
49
|
+
!!x.match(stylesFileRegex) ||
|
|
50
|
+
!!x.match(mimetypeFileRegex) ||
|
|
51
|
+
(!!config.extractAttachments && !!x.match(mediaFileRegex)));
|
|
52
|
+
// 1. Determine File Type
|
|
53
|
+
const mimetypeFile = files.find(f => f.path === 'mimetype');
|
|
54
|
+
let fileType = 'odt'; // Default
|
|
55
|
+
if (mimetypeFile) {
|
|
56
|
+
const mime = mimetypeFile.content.toString().trim();
|
|
57
|
+
if (mime.includes('spreadsheet'))
|
|
58
|
+
fileType = 'ods';
|
|
59
|
+
else if (mime.includes('presentation'))
|
|
60
|
+
fileType = 'odp';
|
|
61
|
+
else if (mime.includes('text'))
|
|
62
|
+
fileType = 'odt';
|
|
63
|
+
}
|
|
64
|
+
const mainContentFile = files.find(f => f.path === 'content.xml') || files.find(f => f.path.match(contentFileRegex));
|
|
65
|
+
const stylesFile = files.find(f => f.path === 'styles.xml');
|
|
66
|
+
const content = [];
|
|
67
|
+
const notes = [];
|
|
68
|
+
// Style Map: styleName -> TextFormatting
|
|
69
|
+
// Inline style parsing (from content.xml automatic styles)
|
|
70
|
+
const styleMap = {};
|
|
71
|
+
const paragraphStyleMap = {};
|
|
72
|
+
const listCounters = {}; // Track item index per listId/level
|
|
73
|
+
// Helper to parse styles
|
|
74
|
+
const parseStyles = (xmlString) => {
|
|
75
|
+
const xml = (0, xmlUtils_1.parseXmlString)(xmlString);
|
|
76
|
+
const styles = (0, xmlUtils_1.getElementsByTagName)(xml, "style:style");
|
|
77
|
+
for (const style of styles) {
|
|
78
|
+
const name = style.getAttribute("style:name");
|
|
79
|
+
if (!name)
|
|
80
|
+
continue;
|
|
81
|
+
const styleInfo = {};
|
|
82
|
+
// Parse paragraph properties for alignment and drop caps
|
|
83
|
+
const paraProps = (0, xmlUtils_1.getElementsByTagName)(style, "style:paragraph-properties")[0];
|
|
84
|
+
if (paraProps) {
|
|
85
|
+
const textAlign = paraProps.getAttribute("fo:text-align");
|
|
86
|
+
if (textAlign) {
|
|
87
|
+
const alignMap = {
|
|
88
|
+
'start': 'left',
|
|
89
|
+
'left': 'left',
|
|
90
|
+
'center': 'center',
|
|
91
|
+
'end': 'right',
|
|
92
|
+
'right': 'right',
|
|
93
|
+
'justify': 'justify'
|
|
94
|
+
};
|
|
95
|
+
if (alignMap[textAlign]) {
|
|
96
|
+
styleInfo.alignment = alignMap[textAlign];
|
|
97
|
+
}
|
|
98
|
+
}
|
|
99
|
+
// Detect Drop Caps
|
|
100
|
+
const dropCap = (0, xmlUtils_1.getElementsByTagName)(paraProps, "style:drop-cap")[0];
|
|
101
|
+
if (dropCap) {
|
|
102
|
+
styleInfo.dropCap = true;
|
|
103
|
+
}
|
|
104
|
+
}
|
|
105
|
+
if (Object.keys(styleInfo).length > 0) {
|
|
106
|
+
paragraphStyleMap[name] = styleInfo;
|
|
107
|
+
}
|
|
108
|
+
// Parse text properties
|
|
109
|
+
const textProps = (0, xmlUtils_1.getElementsByTagName)(style, "style:text-properties")[0];
|
|
110
|
+
// Parse table cell properties (for ODS background)
|
|
111
|
+
const cellProps = (0, xmlUtils_1.getElementsByTagName)(style, "style:table-cell-properties")[0];
|
|
112
|
+
const formatting = {};
|
|
113
|
+
if (cellProps) {
|
|
114
|
+
const bgColor = cellProps.getAttribute("fo:background-color");
|
|
115
|
+
if (bgColor && bgColor !== 'transparent')
|
|
116
|
+
formatting.backgroundColor = bgColor;
|
|
117
|
+
}
|
|
118
|
+
if (textProps) {
|
|
119
|
+
if (textProps.getAttribute("fo:font-weight") === "bold" || textProps.getAttribute("style:font-weight-asian") === "bold")
|
|
120
|
+
formatting.bold = true;
|
|
121
|
+
if (textProps.getAttribute("fo:font-style") === "italic" || textProps.getAttribute("style:font-style-asian") === "italic")
|
|
122
|
+
formatting.italic = true;
|
|
123
|
+
if (textProps.getAttribute("style:text-underline-style") === "solid")
|
|
124
|
+
formatting.underline = true;
|
|
125
|
+
if (textProps.getAttribute("style:text-line-through-style") === "solid")
|
|
126
|
+
formatting.strikethrough = true;
|
|
127
|
+
const size = textProps.getAttribute("fo:font-size") || textProps.getAttribute("style:font-size-asian");
|
|
128
|
+
if (size)
|
|
129
|
+
formatting.size = size;
|
|
130
|
+
const color = textProps.getAttribute("fo:color");
|
|
131
|
+
if (color)
|
|
132
|
+
formatting.color = color;
|
|
133
|
+
// Background color (text level) - override cell level if present?
|
|
134
|
+
const bgColor = textProps.getAttribute("fo:background-color");
|
|
135
|
+
if (bgColor && bgColor !== 'transparent')
|
|
136
|
+
formatting.backgroundColor = bgColor;
|
|
137
|
+
// Font family
|
|
138
|
+
const fontName = textProps.getAttribute("style:font-name") || textProps.getAttribute("fo:font-family");
|
|
139
|
+
if (fontName)
|
|
140
|
+
formatting.font = fontName;
|
|
141
|
+
// Subscript/Superscript from text-position (e.g., "sub 58%" or "super 58%")
|
|
142
|
+
const textPosition = textProps.getAttribute("style:text-position");
|
|
143
|
+
if (textPosition) {
|
|
144
|
+
if (textPosition.startsWith("sub"))
|
|
145
|
+
formatting.subscript = true;
|
|
146
|
+
if (textPosition.startsWith("super"))
|
|
147
|
+
formatting.superscript = true;
|
|
148
|
+
}
|
|
149
|
+
if (Object.keys(formatting).length > 0)
|
|
150
|
+
styleMap[name] = formatting;
|
|
151
|
+
}
|
|
152
|
+
}
|
|
153
|
+
};
|
|
154
|
+
if (stylesFile) {
|
|
155
|
+
parseStyles(stylesFile.content.toString());
|
|
156
|
+
}
|
|
157
|
+
/**
|
|
158
|
+
* Helper to parse a paragraph node (text:p or text:h) and extract its content.
|
|
159
|
+
* Returns the paragraph content without creating a content node.
|
|
160
|
+
*
|
|
161
|
+
* @param node - The paragraph element to parse
|
|
162
|
+
* Helper to parse inline content (text, spans, links, notes, etc.) recursively.
|
|
163
|
+
*
|
|
164
|
+
* @param node - The element to parse (paragraph, span, or link)
|
|
165
|
+
* @param styleMap - Map of style names to formatting
|
|
166
|
+
* @param config - Parser configuration
|
|
167
|
+
* @param notes - Optional array to collect footnotes/endnotes
|
|
168
|
+
* @param paragraphStyleMap - Map of style names to alignments and props (needed for notes)
|
|
169
|
+
* @param parentFormatting - Formatting inherited from parent (e.g. span inside span)
|
|
170
|
+
* @param linkMetadata - Metadata inherited from parent link
|
|
171
|
+
* @returns Object containing text and children
|
|
172
|
+
*/
|
|
173
|
+
const parseInlineContent = (node, styleMap, config, notes, paragraphStyleMap, parentFormatting = {}, linkMetadata) => {
|
|
174
|
+
const children = [];
|
|
175
|
+
let fullText = '';
|
|
176
|
+
if (!node.childNodes)
|
|
177
|
+
return { text: '', children: [] };
|
|
178
|
+
for (let i = 0; i < node.childNodes.length; i++) {
|
|
179
|
+
const child = node.childNodes[i];
|
|
180
|
+
if (child.nodeType === 3) { // Text node
|
|
181
|
+
const text = child.textContent || '';
|
|
182
|
+
if (text) {
|
|
183
|
+
fullText += text;
|
|
184
|
+
children.push({
|
|
185
|
+
type: 'text',
|
|
186
|
+
text: text,
|
|
187
|
+
formatting: parentFormatting,
|
|
188
|
+
metadata: linkMetadata ? { ...linkMetadata } : undefined
|
|
189
|
+
});
|
|
190
|
+
}
|
|
191
|
+
}
|
|
192
|
+
else if (child.nodeType === 1) {
|
|
193
|
+
const element = child;
|
|
194
|
+
const tagName = element.tagName;
|
|
195
|
+
if (tagName === 'text:s') {
|
|
196
|
+
// Space
|
|
197
|
+
const count = parseInt(element.getAttribute('text:c') || '1');
|
|
198
|
+
const spaces = ' '.repeat(count);
|
|
199
|
+
fullText += spaces;
|
|
200
|
+
children.push({
|
|
201
|
+
type: 'text',
|
|
202
|
+
text: spaces,
|
|
203
|
+
formatting: parentFormatting,
|
|
204
|
+
metadata: linkMetadata ? { ...linkMetadata } : undefined
|
|
205
|
+
});
|
|
206
|
+
}
|
|
207
|
+
else if (tagName === 'text:tab') {
|
|
208
|
+
// Tab
|
|
209
|
+
fullText += '\t';
|
|
210
|
+
children.push({
|
|
211
|
+
type: 'text',
|
|
212
|
+
text: '\t',
|
|
213
|
+
formatting: parentFormatting,
|
|
214
|
+
metadata: linkMetadata ? { ...linkMetadata } : undefined
|
|
215
|
+
});
|
|
216
|
+
}
|
|
217
|
+
else if (tagName === 'text:line-break') {
|
|
218
|
+
// Line break
|
|
219
|
+
fullText += '\n';
|
|
220
|
+
children.push({
|
|
221
|
+
type: 'text',
|
|
222
|
+
text: '\n',
|
|
223
|
+
formatting: parentFormatting,
|
|
224
|
+
metadata: linkMetadata ? { ...linkMetadata } : undefined
|
|
225
|
+
});
|
|
226
|
+
}
|
|
227
|
+
else if (tagName === 'text:span') {
|
|
228
|
+
// Formatted text span
|
|
229
|
+
const styleName = element.getAttribute("text:style-name");
|
|
230
|
+
const formatting = styleName ? { ...parentFormatting, ...styleMap[styleName] } : parentFormatting;
|
|
231
|
+
const spanContent = parseInlineContent(element, styleMap, config, notes, paragraphStyleMap, formatting, linkMetadata);
|
|
232
|
+
fullText += spanContent.text;
|
|
233
|
+
children.push(...spanContent.children);
|
|
234
|
+
}
|
|
235
|
+
else if (tagName === 'text:a') {
|
|
236
|
+
// Hyperlink
|
|
237
|
+
const href = element.getAttribute('xlink:href') || '';
|
|
238
|
+
const linkType = href.startsWith('#') ? 'internal' : 'external';
|
|
239
|
+
const newLinkMetadata = { link: href, linkType: linkType };
|
|
240
|
+
const linkContent = parseInlineContent(element, styleMap, config, notes, paragraphStyleMap, parentFormatting, newLinkMetadata);
|
|
241
|
+
fullText += linkContent.text;
|
|
242
|
+
children.push(...linkContent.children);
|
|
243
|
+
}
|
|
244
|
+
else if (tagName === 'text:note' && !config.ignoreNotes) {
|
|
245
|
+
// Footnote or endnote
|
|
246
|
+
const noteClass = (element.getAttribute('text:note-class') || 'footnote');
|
|
247
|
+
const noteId = element.getAttribute('text:id') || element.getAttribute('xml:id') || undefined;
|
|
248
|
+
const noteBody = (0, xmlUtils_1.getElementsByTagName)(element, "text:note-body")[0];
|
|
249
|
+
if (noteBody) {
|
|
250
|
+
// Extract note content recursively
|
|
251
|
+
const notePs = (0, xmlUtils_1.getElementsByTagName)(noteBody, "text:p");
|
|
252
|
+
const noteChildren = [];
|
|
253
|
+
let noteText = '';
|
|
254
|
+
for (const np of notePs) {
|
|
255
|
+
const npContent = parseParagraphContent(np, paragraphStyleMap, styleMap, config);
|
|
256
|
+
noteText += (noteText ? ' ' : '') + npContent.text;
|
|
257
|
+
const npNode = {
|
|
258
|
+
type: 'paragraph',
|
|
259
|
+
text: npContent.text,
|
|
260
|
+
children: npContent.children,
|
|
261
|
+
metadata: npContent.alignment ? { alignment: npContent.alignment } : undefined
|
|
262
|
+
};
|
|
263
|
+
noteChildren.push(npNode);
|
|
264
|
+
}
|
|
265
|
+
const noteNode = {
|
|
266
|
+
type: 'note',
|
|
267
|
+
text: noteText,
|
|
268
|
+
children: noteChildren,
|
|
269
|
+
metadata: {
|
|
270
|
+
noteType: noteClass,
|
|
271
|
+
noteId: noteId
|
|
272
|
+
}
|
|
273
|
+
};
|
|
274
|
+
if (config.putNotesAtLast) {
|
|
275
|
+
notes.push(noteNode);
|
|
276
|
+
}
|
|
277
|
+
else {
|
|
278
|
+
children.push(noteNode);
|
|
279
|
+
}
|
|
280
|
+
}
|
|
281
|
+
}
|
|
282
|
+
else if (tagName === 'draw:frame') {
|
|
283
|
+
// Inline image
|
|
284
|
+
const frame = element;
|
|
285
|
+
// Extract alt text
|
|
286
|
+
let altText = '';
|
|
287
|
+
const svgTitle = (0, xmlUtils_1.getElementsByTagName)(frame, "svg:title")[0];
|
|
288
|
+
const svgDesc = (0, xmlUtils_1.getElementsByTagName)(frame, "svg:desc")[0];
|
|
289
|
+
if (svgTitle && svgTitle.textContent) {
|
|
290
|
+
altText = svgTitle.textContent;
|
|
291
|
+
}
|
|
292
|
+
else if (svgDesc && svgDesc.textContent) {
|
|
293
|
+
altText = svgDesc.textContent;
|
|
294
|
+
}
|
|
295
|
+
// Extract image href
|
|
296
|
+
let imageHref = '';
|
|
297
|
+
const drawImages = (0, xmlUtils_1.getElementsByTagName)(frame, "draw:image");
|
|
298
|
+
if (drawImages.length > 0) {
|
|
299
|
+
imageHref = drawImages[0].getAttribute("xlink:href") || '';
|
|
300
|
+
if (imageHref) {
|
|
301
|
+
const parts = imageHref.split('/');
|
|
302
|
+
imageHref = parts[parts.length - 1];
|
|
303
|
+
}
|
|
304
|
+
}
|
|
305
|
+
const imageNode = {
|
|
306
|
+
type: 'image',
|
|
307
|
+
text: '',
|
|
308
|
+
children: [],
|
|
309
|
+
metadata: {
|
|
310
|
+
attachmentName: imageHref,
|
|
311
|
+
...(altText ? { altText } : {})
|
|
312
|
+
}
|
|
313
|
+
};
|
|
314
|
+
if (config.includeRawContent) {
|
|
315
|
+
imageNode.rawContent = frame.toString();
|
|
316
|
+
}
|
|
317
|
+
children.push(imageNode);
|
|
318
|
+
}
|
|
319
|
+
}
|
|
320
|
+
}
|
|
321
|
+
return { text: fullText, children };
|
|
322
|
+
};
|
|
323
|
+
/**
|
|
324
|
+
* Helper to parse a paragraph node (text:p or text:h) and extract its content.
|
|
325
|
+
* Returns the paragraph content without creating a content node.
|
|
326
|
+
*
|
|
327
|
+
* @param node - The paragraph element to parse
|
|
328
|
+
* @param paraStyleMap - Map of style names to alignments/props
|
|
329
|
+
* @param styleMap - Map of style names to formatting
|
|
330
|
+
* @param config - Parser configuration
|
|
331
|
+
* @returns Object containing text, children, alignment, and style info
|
|
332
|
+
*/
|
|
333
|
+
const parseParagraphContent = (node, paraStyleMap, styleMap, config) => {
|
|
334
|
+
// Get paragraph style for alignment and drop caps
|
|
335
|
+
const paraStyle = node.getAttribute("text:style-name");
|
|
336
|
+
const styleInfo = paraStyle ? paraStyleMap[paraStyle] : undefined;
|
|
337
|
+
const alignment = styleInfo?.alignment;
|
|
338
|
+
const dropCap = styleInfo?.dropCap;
|
|
339
|
+
// Parse content recursively using the new helper
|
|
340
|
+
const content = parseInlineContent(node, styleMap, config, notes, paraStyleMap);
|
|
341
|
+
// Add style name to metadata of children if they don't have one
|
|
342
|
+
if (paraStyle) {
|
|
343
|
+
content.children.forEach(child => {
|
|
344
|
+
if (child.type === 'text') {
|
|
345
|
+
if (!child.metadata)
|
|
346
|
+
child.metadata = {};
|
|
347
|
+
// Only add style if it's a text node and doesn't have one?
|
|
348
|
+
// Or just add it.
|
|
349
|
+
// Cast to any to avoid union type issues for now, or check type
|
|
350
|
+
const meta = child.metadata;
|
|
351
|
+
if (!meta.style)
|
|
352
|
+
meta.style = paraStyle;
|
|
353
|
+
}
|
|
354
|
+
});
|
|
355
|
+
}
|
|
356
|
+
// Fallback: if no children were created but there's text content
|
|
357
|
+
if (content.children.length === 0 && node.textContent) {
|
|
358
|
+
const fullText = node.textContent;
|
|
359
|
+
if (fullText.trim()) {
|
|
360
|
+
content.text = fullText;
|
|
361
|
+
content.children.push({
|
|
362
|
+
type: 'text',
|
|
363
|
+
text: fullText
|
|
364
|
+
});
|
|
365
|
+
}
|
|
366
|
+
}
|
|
367
|
+
// Handle Drop Cap: Apply large font to first letter if configured
|
|
368
|
+
if (dropCap && content.children.length > 0) {
|
|
369
|
+
const firstChild = content.children[0];
|
|
370
|
+
if (firstChild.type === 'text' && firstChild.text) {
|
|
371
|
+
if (firstChild.text.length === 1) {
|
|
372
|
+
// Already a single letter, just apply formatting
|
|
373
|
+
firstChild.formatting = { ...firstChild.formatting, size: '58.5pt' };
|
|
374
|
+
}
|
|
375
|
+
else {
|
|
376
|
+
// Split text node
|
|
377
|
+
const firstChar = firstChild.text[0];
|
|
378
|
+
const restText = firstChild.text.substring(1);
|
|
379
|
+
const dropCapNode = {
|
|
380
|
+
type: 'text',
|
|
381
|
+
text: firstChar,
|
|
382
|
+
formatting: { ...firstChild.formatting, size: '58.5pt' },
|
|
383
|
+
metadata: firstChild.metadata
|
|
384
|
+
};
|
|
385
|
+
// Update original node
|
|
386
|
+
firstChild.text = restText;
|
|
387
|
+
// Insert drop cap node
|
|
388
|
+
content.children.unshift(dropCapNode);
|
|
389
|
+
}
|
|
390
|
+
}
|
|
391
|
+
}
|
|
392
|
+
return { text: content.text, children: content.children, alignment, style: paraStyle || undefined };
|
|
393
|
+
};
|
|
394
|
+
/**
|
|
395
|
+
* Helper to parse a table node and extract its structure.
|
|
396
|
+
* Properly creates table → row → cell hierarchy with metadata.
|
|
397
|
+
*
|
|
398
|
+
* @param tableNode - The table:table element
|
|
399
|
+
* @param paraStyleMap - Map of style names to alignments
|
|
400
|
+
* @param styleMap - Map of style names to formatting
|
|
401
|
+
* @param config - Parser configuration
|
|
402
|
+
* @returns Table content node with proper structure
|
|
403
|
+
*/
|
|
404
|
+
const parseTable = (tableNode, paraStyleMap, styleMap, config) => {
|
|
405
|
+
const rows = [];
|
|
406
|
+
// Use getDirectChildren to avoid nested table rows
|
|
407
|
+
const tableRows = (0, xmlUtils_1.getDirectChildren)(tableNode, "table:table-row");
|
|
408
|
+
let rowIndex = 0;
|
|
409
|
+
for (const row of tableRows) {
|
|
410
|
+
const cells = [];
|
|
411
|
+
// Use getDirectChildren to avoid nested table cells
|
|
412
|
+
const tableCells = (0, xmlUtils_1.getDirectChildren)(row, "table:table-cell");
|
|
413
|
+
const rowsRepeated = parseInt(row.getAttribute("table:number-rows-repeated") || "1");
|
|
414
|
+
let colIndex = 0;
|
|
415
|
+
for (const cell of tableCells) {
|
|
416
|
+
const cellChildren = [];
|
|
417
|
+
let cellTextRef = { value: '' };
|
|
418
|
+
const colsRepeated = parseInt(cell.getAttribute("table:number-columns-repeated") || "1");
|
|
419
|
+
const colSpan = parseInt(cell.getAttribute("table:number-columns-spanned") || "1");
|
|
420
|
+
const rowSpan = parseInt(cell.getAttribute("table:number-rows-spanned") || "1");
|
|
421
|
+
// Helper to recursively process cell children (handles frames, text-boxes, etc. in ODP)
|
|
422
|
+
const processChildren = (node) => {
|
|
423
|
+
if (!node.childNodes)
|
|
424
|
+
return;
|
|
425
|
+
for (let i = 0; i < node.childNodes.length; i++) {
|
|
426
|
+
const child = node.childNodes[i];
|
|
427
|
+
if (child.nodeType === 1) { // Element
|
|
428
|
+
const element = child;
|
|
429
|
+
if (element.tagName === "text:p" || element.tagName === "text:h") {
|
|
430
|
+
const pContent = parseParagraphContent(element, paraStyleMap, styleMap, config);
|
|
431
|
+
const pNode = {
|
|
432
|
+
type: element.tagName === "text:h" ? 'heading' : 'paragraph',
|
|
433
|
+
text: pContent.text,
|
|
434
|
+
children: pContent.children,
|
|
435
|
+
metadata: {
|
|
436
|
+
...(pContent.alignment ? { alignment: pContent.alignment } : {}),
|
|
437
|
+
...(pContent.style ? { style: pContent.style } : {})
|
|
438
|
+
}
|
|
439
|
+
};
|
|
440
|
+
// Clean up metadata if empty
|
|
441
|
+
if (Object.keys(pNode.metadata || {}).length === 0)
|
|
442
|
+
delete pNode.metadata;
|
|
443
|
+
if (element.tagName === "text:h") {
|
|
444
|
+
if (!pNode.metadata)
|
|
445
|
+
pNode.metadata = {};
|
|
446
|
+
pNode.metadata.level = parseInt(element.getAttribute("text:outline-level") || "1");
|
|
447
|
+
}
|
|
448
|
+
if (config.includeRawContent) {
|
|
449
|
+
pNode.rawContent = element.toString();
|
|
450
|
+
}
|
|
451
|
+
cellChildren.push(pNode);
|
|
452
|
+
cellTextRef.value += pContent.text;
|
|
453
|
+
// Add newline if there are multiple paragraphs/headings
|
|
454
|
+
if (cellTextRef.value && !cellTextRef.value.endsWith('\n')) {
|
|
455
|
+
cellTextRef.value += '\n';
|
|
456
|
+
}
|
|
457
|
+
}
|
|
458
|
+
else if (element.tagName === "table:table") {
|
|
459
|
+
// Recursive call for nested table
|
|
460
|
+
const nestedTableNode = parseTable(element, paraStyleMap, styleMap, config);
|
|
461
|
+
cellChildren.push(nestedTableNode);
|
|
462
|
+
}
|
|
463
|
+
else if (element.tagName === "draw:frame" || element.tagName === "draw:text-box") {
|
|
464
|
+
// Recursively process container content (common in ODP)
|
|
465
|
+
processChildren(element);
|
|
466
|
+
}
|
|
467
|
+
}
|
|
468
|
+
}
|
|
469
|
+
};
|
|
470
|
+
processChildren(cell);
|
|
471
|
+
let cellText = cellTextRef.value;
|
|
472
|
+
// Trim trailing newline from cellText
|
|
473
|
+
if (cellText.endsWith('\n')) {
|
|
474
|
+
cellText = cellText.slice(0, -1);
|
|
475
|
+
}
|
|
476
|
+
// Add cell(s) for repeated columns
|
|
477
|
+
for (let k = 0; k < colsRepeated; k++) {
|
|
478
|
+
const cellNode = {
|
|
479
|
+
type: 'cell',
|
|
480
|
+
text: cellText,
|
|
481
|
+
children: cellChildren.length > 0 ? (k === 0 ? cellChildren : JSON.parse(JSON.stringify(cellChildren))) : [],
|
|
482
|
+
metadata: { row: rowIndex, col: colIndex }
|
|
483
|
+
};
|
|
484
|
+
const cellMetadata = cellNode.metadata;
|
|
485
|
+
if (colSpan > 1)
|
|
486
|
+
cellMetadata.colSpan = colSpan;
|
|
487
|
+
if (rowSpan > 1)
|
|
488
|
+
cellMetadata.rowSpan = rowSpan;
|
|
489
|
+
if (config.includeRawContent) {
|
|
490
|
+
cellNode.rawContent = cell.toString();
|
|
491
|
+
}
|
|
492
|
+
cells.push(cellNode);
|
|
493
|
+
colIndex++;
|
|
494
|
+
}
|
|
495
|
+
}
|
|
496
|
+
// Add row(s) for repeated rows
|
|
497
|
+
for (let k = 0; k < rowsRepeated; k++) {
|
|
498
|
+
const rowNode = {
|
|
499
|
+
type: 'row',
|
|
500
|
+
children: k === 0 ? cells : JSON.parse(JSON.stringify(cells))
|
|
501
|
+
};
|
|
502
|
+
// Fix row indices for repeated rows
|
|
503
|
+
if (k > 0) {
|
|
504
|
+
rowNode.children?.forEach(c => {
|
|
505
|
+
if (c.metadata && 'row' in c.metadata) {
|
|
506
|
+
c.metadata.row = rowIndex;
|
|
507
|
+
}
|
|
508
|
+
});
|
|
509
|
+
}
|
|
510
|
+
if (config.includeRawContent) {
|
|
511
|
+
rowNode.rawContent = row.toString();
|
|
512
|
+
}
|
|
513
|
+
rows.push(rowNode);
|
|
514
|
+
rowIndex++;
|
|
515
|
+
}
|
|
516
|
+
}
|
|
517
|
+
return {
|
|
518
|
+
type: 'table',
|
|
519
|
+
children: rows
|
|
520
|
+
};
|
|
521
|
+
};
|
|
522
|
+
const parseContentXml = (xmlString) => {
|
|
523
|
+
const xml = (0, xmlUtils_1.parseXmlString)(xmlString);
|
|
524
|
+
const body = (0, xmlUtils_1.getElementsByTagName)(xml, "office:body")[0];
|
|
525
|
+
// Parse automatic styles (local to content.xml)
|
|
526
|
+
const automaticStyles = (0, xmlUtils_1.getElementsByTagName)(xml, "office:automatic-styles")[0];
|
|
527
|
+
if (automaticStyles) {
|
|
528
|
+
const styles = (0, xmlUtils_1.getElementsByTagName)(automaticStyles, "style:style");
|
|
529
|
+
for (const style of styles) {
|
|
530
|
+
const name = style.getAttribute("style:name");
|
|
531
|
+
if (!name)
|
|
532
|
+
continue;
|
|
533
|
+
// Parse paragraph properties for alignment
|
|
534
|
+
const paraProps = (0, xmlUtils_1.getElementsByTagName)(style, "style:paragraph-properties")[0];
|
|
535
|
+
const styleInfo = {};
|
|
536
|
+
if (paraProps) {
|
|
537
|
+
const textAlign = paraProps.getAttribute("fo:text-align");
|
|
538
|
+
if (textAlign) {
|
|
539
|
+
const alignMap = {
|
|
540
|
+
'start': 'left',
|
|
541
|
+
'left': 'left',
|
|
542
|
+
'center': 'center',
|
|
543
|
+
'end': 'right',
|
|
544
|
+
'right': 'right',
|
|
545
|
+
'justify': 'justify'
|
|
546
|
+
};
|
|
547
|
+
if (alignMap[textAlign]) {
|
|
548
|
+
styleInfo.alignment = alignMap[textAlign];
|
|
549
|
+
}
|
|
550
|
+
}
|
|
551
|
+
const dropCap = (0, xmlUtils_1.getElementsByTagName)(paraProps, "style:drop-cap")[0];
|
|
552
|
+
if (dropCap)
|
|
553
|
+
styleInfo.dropCap = true;
|
|
554
|
+
}
|
|
555
|
+
if (Object.keys(styleInfo).length > 0) {
|
|
556
|
+
paragraphStyleMap[name] = styleInfo;
|
|
557
|
+
}
|
|
558
|
+
const textProps = (0, xmlUtils_1.getElementsByTagName)(style, "style:text-properties")[0];
|
|
559
|
+
if (textProps) {
|
|
560
|
+
const formatting = {};
|
|
561
|
+
if (textProps.getAttribute("fo:font-weight") === "bold" || textProps.getAttribute("style:font-weight-asian") === "bold")
|
|
562
|
+
formatting.bold = true;
|
|
563
|
+
if (textProps.getAttribute("fo:font-style") === "italic" || textProps.getAttribute("style:font-style-asian") === "italic")
|
|
564
|
+
formatting.italic = true;
|
|
565
|
+
if (textProps.getAttribute("style:text-underline-style") === "solid")
|
|
566
|
+
formatting.underline = true;
|
|
567
|
+
if (textProps.getAttribute("style:text-line-through-style") === "solid")
|
|
568
|
+
formatting.strikethrough = true;
|
|
569
|
+
const size = textProps.getAttribute("fo:font-size") || textProps.getAttribute("style:font-size-asian");
|
|
570
|
+
if (size)
|
|
571
|
+
formatting.size = size;
|
|
572
|
+
const color = textProps.getAttribute("fo:color");
|
|
573
|
+
if (color)
|
|
574
|
+
formatting.color = color;
|
|
575
|
+
// Background color
|
|
576
|
+
const bgColor = textProps.getAttribute("fo:background-color");
|
|
577
|
+
if (bgColor && bgColor !== 'transparent')
|
|
578
|
+
formatting.backgroundColor = bgColor;
|
|
579
|
+
// Font family
|
|
580
|
+
const fontName = textProps.getAttribute("style:font-name") || textProps.getAttribute("fo:font-family");
|
|
581
|
+
if (fontName)
|
|
582
|
+
formatting.font = fontName;
|
|
583
|
+
// Subscript/Superscript from text-position (e.g., "sub 58%" or "super 58%")
|
|
584
|
+
const textPosition = textProps.getAttribute("style:text-position");
|
|
585
|
+
if (textPosition) {
|
|
586
|
+
if (textPosition.startsWith("sub"))
|
|
587
|
+
formatting.subscript = true;
|
|
588
|
+
if (textPosition.startsWith("super"))
|
|
589
|
+
formatting.superscript = true;
|
|
590
|
+
}
|
|
591
|
+
if (Object.keys(formatting).length > 0)
|
|
592
|
+
styleMap[name] = formatting;
|
|
593
|
+
}
|
|
594
|
+
}
|
|
595
|
+
}
|
|
596
|
+
/**
|
|
597
|
+
* Recursively traverses a node and its children to extract content.
|
|
598
|
+
* Properly handles paragraphs, headings, tables, lists, and frames.
|
|
599
|
+
*
|
|
600
|
+
* @param node - The element to traverse
|
|
601
|
+
* @param targetArray - The array to push extracted content nodes to
|
|
602
|
+
* @param forceHeading - If true, treats all paragraphs as headings (used for slide titles)
|
|
603
|
+
*/
|
|
604
|
+
const traverse = (node, targetArray, forceHeading = false) => {
|
|
605
|
+
if (node.tagName === "text:p") {
|
|
606
|
+
const pContent = parseParagraphContent(node, paragraphStyleMap, styleMap, config);
|
|
607
|
+
const type = (forceHeading || (node.getAttribute("text:style-name") || '').toLowerCase().includes('title')) ? 'heading' : 'paragraph';
|
|
608
|
+
const pNode = {
|
|
609
|
+
type,
|
|
610
|
+
text: pContent.text,
|
|
611
|
+
children: pContent.children,
|
|
612
|
+
metadata: {
|
|
613
|
+
...(pContent.alignment ? { alignment: pContent.alignment } : {}),
|
|
614
|
+
...(pContent.style ? { style: pContent.style } : {})
|
|
615
|
+
}
|
|
616
|
+
};
|
|
617
|
+
if (type === 'heading' && pNode.metadata) {
|
|
618
|
+
pNode.metadata.level = pNode.metadata.level || 1;
|
|
619
|
+
}
|
|
620
|
+
// Clean up metadata if empty
|
|
621
|
+
if (Object.keys(pNode.metadata || {}).length === 0)
|
|
622
|
+
delete pNode.metadata;
|
|
623
|
+
if (config.includeRawContent) {
|
|
624
|
+
pNode.rawContent = node.toString();
|
|
625
|
+
}
|
|
626
|
+
targetArray.push(pNode);
|
|
627
|
+
}
|
|
628
|
+
else if (node.tagName === "text:h") {
|
|
629
|
+
const level = parseInt(node.getAttribute("text:outline-level") || "1");
|
|
630
|
+
const hContent = parseParagraphContent(node, paragraphStyleMap, styleMap, config);
|
|
631
|
+
const hNode = {
|
|
632
|
+
type: 'heading',
|
|
633
|
+
text: hContent.text,
|
|
634
|
+
children: hContent.children,
|
|
635
|
+
metadata: {
|
|
636
|
+
level,
|
|
637
|
+
...(hContent.alignment ? { alignment: hContent.alignment } : {}),
|
|
638
|
+
...(hContent.style ? { style: hContent.style } : {})
|
|
639
|
+
}
|
|
640
|
+
};
|
|
641
|
+
if (config.includeRawContent) {
|
|
642
|
+
hNode.rawContent = node.toString();
|
|
643
|
+
}
|
|
644
|
+
targetArray.push(hNode);
|
|
645
|
+
}
|
|
646
|
+
else if (node.tagName === "table:table") {
|
|
647
|
+
// Parse table with proper structure
|
|
648
|
+
const tableNode = parseTable(node, paragraphStyleMap, styleMap, config);
|
|
649
|
+
if (config.includeRawContent) {
|
|
650
|
+
tableNode.rawContent = node.toString();
|
|
651
|
+
}
|
|
652
|
+
targetArray.push(tableNode);
|
|
653
|
+
}
|
|
654
|
+
else if (node.tagName === "text:list") {
|
|
655
|
+
// Parse list structure with proper listId tracking
|
|
656
|
+
const listItems = (0, xmlUtils_1.getDirectChildren)(node, "text:list-item");
|
|
657
|
+
// Get list style name to use as listId (or generate one)
|
|
658
|
+
const listStyleName = node.getAttribute("text:style-name") || node.getAttribute("xml:id");
|
|
659
|
+
const listId = listStyleName || `list-${targetArray.length}`;
|
|
660
|
+
// Determine list type by checking the list style definition
|
|
661
|
+
let listType = 'unordered';
|
|
662
|
+
let isVisible = false;
|
|
663
|
+
let styleNameToCheck = listStyleName;
|
|
664
|
+
// If no style name, check parent list for inherited style
|
|
665
|
+
if (!styleNameToCheck) {
|
|
666
|
+
let parentNode = node.parentNode;
|
|
667
|
+
while (parentNode && !styleNameToCheck) {
|
|
668
|
+
if (parentNode.nodeName === 'text:list') {
|
|
669
|
+
styleNameToCheck = parentNode.getAttribute("text:style-name");
|
|
670
|
+
if (styleNameToCheck)
|
|
671
|
+
break;
|
|
672
|
+
}
|
|
673
|
+
parentNode = parentNode.parentNode;
|
|
674
|
+
}
|
|
675
|
+
}
|
|
676
|
+
// Try to find list style in automatic styles to determine type and visibility
|
|
677
|
+
if (styleNameToCheck) {
|
|
678
|
+
const automaticStyles = (0, xmlUtils_1.getElementsByTagName)((0, xmlUtils_1.parseXmlString)(mainContentFile?.content.toString() || ''), "office:automatic-styles")[0];
|
|
679
|
+
if (automaticStyles) {
|
|
680
|
+
const listStyles = (0, xmlUtils_1.getElementsByTagName)(automaticStyles, "text:list-style");
|
|
681
|
+
for (const listStyle of listStyles) {
|
|
682
|
+
if (listStyle.getAttribute("style:name") === styleNameToCheck) {
|
|
683
|
+
// Check if it has bullet or number level styles
|
|
684
|
+
const bulletLevels = (0, xmlUtils_1.getElementsByTagName)(listStyle, "text:list-level-style-bullet");
|
|
685
|
+
const numberLevels = (0, xmlUtils_1.getElementsByTagName)(listStyle, "text:list-level-style-number");
|
|
686
|
+
const imageLevels = (0, xmlUtils_1.getElementsByTagName)(listStyle, "text:list-level-style-image");
|
|
687
|
+
if (numberLevels.length > 0) {
|
|
688
|
+
listType = 'ordered';
|
|
689
|
+
isVisible = numberLevels.some(l => !!l.getAttribute("style:num-format"));
|
|
690
|
+
}
|
|
691
|
+
else if (bulletLevels.length > 0) {
|
|
692
|
+
listType = 'unordered';
|
|
693
|
+
isVisible = bulletLevels.some(l => !!l.getAttribute("text:bullet-char"));
|
|
694
|
+
}
|
|
695
|
+
if (imageLevels.length > 0)
|
|
696
|
+
isVisible = true;
|
|
697
|
+
break;
|
|
698
|
+
}
|
|
699
|
+
}
|
|
700
|
+
}
|
|
701
|
+
// Also check in styles.xml if still unordered and hidden
|
|
702
|
+
if (stylesFile && !isVisible) {
|
|
703
|
+
const stylesXml = (0, xmlUtils_1.parseXmlString)(stylesFile.content.toString());
|
|
704
|
+
const listStyles = (0, xmlUtils_1.getElementsByTagName)(stylesXml, "text:list-style");
|
|
705
|
+
for (const listStyle of listStyles) {
|
|
706
|
+
if (listStyle.getAttribute("style:name") === styleNameToCheck) {
|
|
707
|
+
const bulletLevels = (0, xmlUtils_1.getElementsByTagName)(listStyle, "text:list-level-style-bullet");
|
|
708
|
+
const numberLevels = (0, xmlUtils_1.getElementsByTagName)(listStyle, "text:list-level-style-number");
|
|
709
|
+
const imageLevels = (0, xmlUtils_1.getElementsByTagName)(listStyle, "text:list-level-style-image");
|
|
710
|
+
if (numberLevels.length > 0) {
|
|
711
|
+
listType = 'ordered';
|
|
712
|
+
isVisible = numberLevels.some(l => !!l.getAttribute("style:num-format"));
|
|
713
|
+
}
|
|
714
|
+
else if (bulletLevels.length > 0) {
|
|
715
|
+
listType = 'unordered';
|
|
716
|
+
isVisible = bulletLevels.some(l => !!l.getAttribute("text:bullet-char"));
|
|
717
|
+
}
|
|
718
|
+
if (imageLevels.length > 0)
|
|
719
|
+
isVisible = true;
|
|
720
|
+
break;
|
|
721
|
+
}
|
|
722
|
+
}
|
|
723
|
+
}
|
|
724
|
+
}
|
|
725
|
+
// If the list is not visible, it's likely a layout list used by Impress.
|
|
726
|
+
// We should traverse its items and treat their content as regular nodes.
|
|
727
|
+
if (!isVisible) {
|
|
728
|
+
for (let i = 0; i < listItems.length; i++) {
|
|
729
|
+
const item = listItems[i];
|
|
730
|
+
if (item.childNodes) {
|
|
731
|
+
for (let j = 0; j < item.childNodes.length; j++) {
|
|
732
|
+
const child = item.childNodes[j];
|
|
733
|
+
if (child.nodeType === 1) { // Element
|
|
734
|
+
traverse(child, targetArray, forceHeading);
|
|
735
|
+
}
|
|
736
|
+
}
|
|
737
|
+
}
|
|
738
|
+
}
|
|
739
|
+
return;
|
|
740
|
+
}
|
|
741
|
+
// Calculate indentation level by counting parent text:list elements
|
|
742
|
+
let indentation = 0;
|
|
743
|
+
let parent = node.parentNode;
|
|
744
|
+
while (parent) {
|
|
745
|
+
if (parent.nodeName === 'text:list') {
|
|
746
|
+
indentation++;
|
|
747
|
+
}
|
|
748
|
+
parent = parent.parentNode;
|
|
749
|
+
}
|
|
750
|
+
// Track list counters for this listId (similar to WordParser)
|
|
751
|
+
if (!listCounters[listId]) {
|
|
752
|
+
listCounters[listId] = {};
|
|
753
|
+
}
|
|
754
|
+
const indentKey = indentation.toString();
|
|
755
|
+
if (listCounters[listId][indentKey] === undefined) {
|
|
756
|
+
listCounters[listId][indentKey] = -1; // Will increment to 0 on first item
|
|
757
|
+
}
|
|
758
|
+
// Process each list item
|
|
759
|
+
for (let i = 0; i < listItems.length; i++) {
|
|
760
|
+
const item = listItems[i];
|
|
761
|
+
// Increment item index for this list/level
|
|
762
|
+
listCounters[listId][indentKey]++;
|
|
763
|
+
const itemIndex = listCounters[listId][indentKey];
|
|
764
|
+
// Reset deeper levels when we encounter an item at this level
|
|
765
|
+
for (let k = indentation + 1; k < 10; k++) {
|
|
766
|
+
if (listCounters[listId][k.toString()] !== undefined) {
|
|
767
|
+
listCounters[listId][k.toString()] = -1;
|
|
768
|
+
}
|
|
769
|
+
}
|
|
770
|
+
// Iterate over direct children of list item (paragraphs, headings, nested lists)
|
|
771
|
+
if (item.childNodes) {
|
|
772
|
+
for (let j = 0; j < item.childNodes.length; j++) {
|
|
773
|
+
const child = item.childNodes[j];
|
|
774
|
+
if (child.nodeType === 1) { // Element
|
|
775
|
+
const element = child;
|
|
776
|
+
if (element.tagName === "text:p") {
|
|
777
|
+
const pContent = parseParagraphContent(element, paragraphStyleMap, styleMap, config);
|
|
778
|
+
const listNode = {
|
|
779
|
+
type: 'list',
|
|
780
|
+
text: pContent.text,
|
|
781
|
+
children: pContent.children,
|
|
782
|
+
metadata: {
|
|
783
|
+
listType,
|
|
784
|
+
indentation,
|
|
785
|
+
itemIndex,
|
|
786
|
+
listId,
|
|
787
|
+
alignment: pContent.alignment || 'left',
|
|
788
|
+
style: pContent.style
|
|
789
|
+
}
|
|
790
|
+
};
|
|
791
|
+
if (config.includeRawContent)
|
|
792
|
+
listNode.rawContent = element.toString();
|
|
793
|
+
targetArray.push(listNode);
|
|
794
|
+
}
|
|
795
|
+
else if (element.tagName === "text:h") {
|
|
796
|
+
const level = parseInt(element.getAttribute("text:outline-level") || "1");
|
|
797
|
+
const hContent = parseParagraphContent(element, paragraphStyleMap, styleMap, config);
|
|
798
|
+
const listNode = {
|
|
799
|
+
type: 'list',
|
|
800
|
+
text: hContent.text,
|
|
801
|
+
children: hContent.children,
|
|
802
|
+
metadata: {
|
|
803
|
+
listType,
|
|
804
|
+
indentation,
|
|
805
|
+
itemIndex,
|
|
806
|
+
listId,
|
|
807
|
+
...(hContent.alignment ? { alignment: hContent.alignment } : {}),
|
|
808
|
+
style: hContent.style
|
|
809
|
+
}
|
|
810
|
+
};
|
|
811
|
+
if (config.includeRawContent)
|
|
812
|
+
listNode.rawContent = element.toString();
|
|
813
|
+
targetArray.push(listNode);
|
|
814
|
+
}
|
|
815
|
+
else if (element.tagName === "text:list") {
|
|
816
|
+
// Recursive call for nested list
|
|
817
|
+
traverse(element, targetArray, forceHeading);
|
|
818
|
+
}
|
|
819
|
+
}
|
|
820
|
+
}
|
|
821
|
+
}
|
|
822
|
+
}
|
|
823
|
+
}
|
|
824
|
+
else if (node.tagName === "draw:frame") {
|
|
825
|
+
const presClass = node.getAttribute("presentation:class");
|
|
826
|
+
const isHeading = presClass === "title" || presClass === "sub-title";
|
|
827
|
+
// In presentations, frames often contain text-boxes, images, tables, or objects
|
|
828
|
+
const textBox = (0, xmlUtils_1.getElementsByTagName)(node, "draw:text-box")[0];
|
|
829
|
+
const image = (0, xmlUtils_1.getElementsByTagName)(node, "draw:image")[0];
|
|
830
|
+
const table = (0, xmlUtils_1.getElementsByTagName)(node, "table:table")[0];
|
|
831
|
+
const object = (0, xmlUtils_1.getElementsByTagName)(node, "draw:object")[0];
|
|
832
|
+
if (textBox) {
|
|
833
|
+
traverse(textBox, targetArray, isHeading || forceHeading);
|
|
834
|
+
}
|
|
835
|
+
else if (table) {
|
|
836
|
+
const tableNode = parseTable(table, paragraphStyleMap, styleMap, config);
|
|
837
|
+
if (config.includeRawContent)
|
|
838
|
+
tableNode.rawContent = table.toString();
|
|
839
|
+
targetArray.push(tableNode);
|
|
840
|
+
}
|
|
841
|
+
else if (image) {
|
|
842
|
+
// Extract alt text from svg:title or svg:desc
|
|
843
|
+
let altText = '';
|
|
844
|
+
const svgTitle = (0, xmlUtils_1.getElementsByTagName)(node, "svg:title")[0];
|
|
845
|
+
const svgDesc = (0, xmlUtils_1.getElementsByTagName)(node, "svg:desc")[0];
|
|
846
|
+
if (svgTitle && svgTitle.textContent) {
|
|
847
|
+
altText = svgTitle.textContent;
|
|
848
|
+
}
|
|
849
|
+
else if (svgDesc && svgDesc.textContent) {
|
|
850
|
+
altText = svgDesc.textContent;
|
|
851
|
+
}
|
|
852
|
+
// Extract image href to link to attachment
|
|
853
|
+
let imageHref = image.getAttribute("xlink:href") || '';
|
|
854
|
+
if (imageHref) {
|
|
855
|
+
const parts = imageHref.split('/');
|
|
856
|
+
imageHref = parts[parts.length - 1];
|
|
857
|
+
}
|
|
858
|
+
const imageNode = {
|
|
859
|
+
type: 'image',
|
|
860
|
+
text: '',
|
|
861
|
+
children: [],
|
|
862
|
+
metadata: {
|
|
863
|
+
attachmentName: imageHref,
|
|
864
|
+
...(altText ? { altText } : {})
|
|
865
|
+
}
|
|
866
|
+
};
|
|
867
|
+
if (config.includeRawContent) {
|
|
868
|
+
imageNode.rawContent = node.toString();
|
|
869
|
+
}
|
|
870
|
+
targetArray.push(imageNode);
|
|
871
|
+
}
|
|
872
|
+
else if (object) {
|
|
873
|
+
// Handle embedded objects like charts
|
|
874
|
+
const href = object.getAttribute("xlink:href");
|
|
875
|
+
if (href) {
|
|
876
|
+
const attachmentName = href.split('/')[0];
|
|
877
|
+
const objectPath = `${attachmentName}/content.xml`;
|
|
878
|
+
const objectFile = files.find(f => f.path === objectPath || f.path.endsWith(objectPath));
|
|
879
|
+
if (objectFile) {
|
|
880
|
+
const chartData = (0, chartUtils_1.extractChartData)(objectFile.content);
|
|
881
|
+
const chartNode = {
|
|
882
|
+
type: 'chart',
|
|
883
|
+
text: chartData.rawTexts.join(" "),
|
|
884
|
+
metadata: {
|
|
885
|
+
attachmentName: attachmentName,
|
|
886
|
+
chartData
|
|
887
|
+
}
|
|
888
|
+
};
|
|
889
|
+
if (config.includeRawContent)
|
|
890
|
+
chartNode.rawContent = node.toString();
|
|
891
|
+
targetArray.push(chartNode);
|
|
892
|
+
}
|
|
893
|
+
else {
|
|
894
|
+
const chartNode = {
|
|
895
|
+
type: 'chart',
|
|
896
|
+
text: "",
|
|
897
|
+
metadata: { attachmentName: attachmentName }
|
|
898
|
+
};
|
|
899
|
+
if (config.includeRawContent)
|
|
900
|
+
chartNode.rawContent = node.toString();
|
|
901
|
+
targetArray.push(chartNode);
|
|
902
|
+
}
|
|
903
|
+
}
|
|
904
|
+
}
|
|
905
|
+
}
|
|
906
|
+
else {
|
|
907
|
+
if (node.childNodes) {
|
|
908
|
+
for (let i = 0; i < node.childNodes.length; i++) {
|
|
909
|
+
const child = node.childNodes[i];
|
|
910
|
+
if (child.nodeType === 1) { // Element
|
|
911
|
+
traverse(child, targetArray, forceHeading);
|
|
912
|
+
}
|
|
913
|
+
}
|
|
914
|
+
}
|
|
915
|
+
}
|
|
916
|
+
};
|
|
917
|
+
// ODS: Spreadsheet
|
|
918
|
+
if (fileType === 'ods') {
|
|
919
|
+
const spreadsheet = (0, xmlUtils_1.getElementsByTagName)(body, "office:spreadsheet")[0];
|
|
920
|
+
if (spreadsheet) {
|
|
921
|
+
const tables = (0, xmlUtils_1.getElementsByTagName)(spreadsheet, "table:table");
|
|
922
|
+
for (let i = 0; i < tables.length; i++) {
|
|
923
|
+
const table = tables[i];
|
|
924
|
+
const sheetName = table.getAttribute("table:name") || `Sheet${i + 1}`;
|
|
925
|
+
const rows = [];
|
|
926
|
+
const tableRows = (0, xmlUtils_1.getElementsByTagName)(table, "table:table-row");
|
|
927
|
+
let rowIndex = 0;
|
|
928
|
+
for (let r = 0; r < tableRows.length; r++) {
|
|
929
|
+
const row = tableRows[r];
|
|
930
|
+
const cells = [];
|
|
931
|
+
const tableCells = (0, xmlUtils_1.getElementsByTagName)(row, "table:table-cell");
|
|
932
|
+
let colIndex = 0;
|
|
933
|
+
const rowsRepeated = parseInt(row.getAttribute("table:number-rows-repeated") || "1");
|
|
934
|
+
for (let c = 0; c < tableCells.length; c++) {
|
|
935
|
+
const cell = tableCells[c];
|
|
936
|
+
const colsRepeated = parseInt(cell.getAttribute("table:number-columns-repeated") || "1");
|
|
937
|
+
// Extract text from cell (paragraphs inside cell)
|
|
938
|
+
let cellText = "";
|
|
939
|
+
const children = [];
|
|
940
|
+
const ps = (0, xmlUtils_1.getElementsByTagName)(cell, "text:p");
|
|
941
|
+
for (let p = 0; p < ps.length; p++) {
|
|
942
|
+
const para = ps[p];
|
|
943
|
+
// Parse text:span elements for formatted text
|
|
944
|
+
const spans = (0, xmlUtils_1.getElementsByTagName)(para, "text:span");
|
|
945
|
+
if (spans.length > 0) {
|
|
946
|
+
for (const span of spans) {
|
|
947
|
+
const styleName = span.getAttribute("text:style-name");
|
|
948
|
+
const formatting = styleName ? styleMap[styleName] : {};
|
|
949
|
+
const text = span.textContent || '';
|
|
950
|
+
cellText += text;
|
|
951
|
+
const textNode = {
|
|
952
|
+
type: 'text',
|
|
953
|
+
text: text,
|
|
954
|
+
formatting: formatting
|
|
955
|
+
};
|
|
956
|
+
children.push(textNode);
|
|
957
|
+
}
|
|
958
|
+
}
|
|
959
|
+
else {
|
|
960
|
+
// No spans - just direct text content
|
|
961
|
+
const text = para.textContent || '';
|
|
962
|
+
cellText += text;
|
|
963
|
+
if (text.trim()) {
|
|
964
|
+
const textNode = {
|
|
965
|
+
type: 'text',
|
|
966
|
+
text: text,
|
|
967
|
+
formatting: {}
|
|
968
|
+
};
|
|
969
|
+
children.push(textNode);
|
|
970
|
+
}
|
|
971
|
+
}
|
|
972
|
+
if (p < ps.length - 1)
|
|
973
|
+
cellText += "\n";
|
|
974
|
+
}
|
|
975
|
+
// Check for embedded draw:frame (images) in cell
|
|
976
|
+
const drawFrames = (0, xmlUtils_1.getElementsByTagName)(cell, "draw:frame");
|
|
977
|
+
for (const frame of drawFrames) {
|
|
978
|
+
// Extract alt text from svg:title or svg:desc
|
|
979
|
+
let altText = '';
|
|
980
|
+
const svgTitle = (0, xmlUtils_1.getElementsByTagName)(frame, "svg:title")[0];
|
|
981
|
+
const svgDesc = (0, xmlUtils_1.getElementsByTagName)(frame, "svg:desc")[0];
|
|
982
|
+
if (svgTitle && svgTitle.textContent) {
|
|
983
|
+
altText = svgTitle.textContent;
|
|
984
|
+
}
|
|
985
|
+
else if (svgDesc && svgDesc.textContent) {
|
|
986
|
+
altText = svgDesc.textContent;
|
|
987
|
+
}
|
|
988
|
+
// Extract image href
|
|
989
|
+
let imageHref = '';
|
|
990
|
+
const drawImages = (0, xmlUtils_1.getElementsByTagName)(frame, "draw:image");
|
|
991
|
+
if (drawImages.length > 0) {
|
|
992
|
+
const rawHref = drawImages[0].getAttribute("xlink:href");
|
|
993
|
+
if (rawHref) {
|
|
994
|
+
const parts = rawHref.split('/');
|
|
995
|
+
imageHref = parts[parts.length - 1];
|
|
996
|
+
}
|
|
997
|
+
}
|
|
998
|
+
// Extract chart object href
|
|
999
|
+
let chartHref = '';
|
|
1000
|
+
const drawObjects = (0, xmlUtils_1.getElementsByTagName)(frame, "draw:object");
|
|
1001
|
+
if (drawObjects.length > 0) {
|
|
1002
|
+
const href = drawObjects[0].getAttribute("xlink:href");
|
|
1003
|
+
if (href) {
|
|
1004
|
+
// Object href is usually "./Object 1"
|
|
1005
|
+
chartHref = href.split('/')[0];
|
|
1006
|
+
}
|
|
1007
|
+
}
|
|
1008
|
+
if (drawImages.length > 0) {
|
|
1009
|
+
// logic for image node
|
|
1010
|
+
const imageNode = {
|
|
1011
|
+
type: 'image',
|
|
1012
|
+
text: '',
|
|
1013
|
+
children: [],
|
|
1014
|
+
metadata: {
|
|
1015
|
+
attachmentName: imageHref,
|
|
1016
|
+
...(altText ? { altText } : {})
|
|
1017
|
+
}
|
|
1018
|
+
};
|
|
1019
|
+
if (config.includeRawContent) {
|
|
1020
|
+
imageNode.rawContent = frame.toString();
|
|
1021
|
+
}
|
|
1022
|
+
children.push(imageNode);
|
|
1023
|
+
}
|
|
1024
|
+
else if (chartHref) {
|
|
1025
|
+
const chartNode = {
|
|
1026
|
+
type: 'chart',
|
|
1027
|
+
text: '',
|
|
1028
|
+
children: [],
|
|
1029
|
+
metadata: {
|
|
1030
|
+
attachmentName: chartHref
|
|
1031
|
+
}
|
|
1032
|
+
};
|
|
1033
|
+
children.push(chartNode);
|
|
1034
|
+
}
|
|
1035
|
+
}
|
|
1036
|
+
// Add cell(s)
|
|
1037
|
+
for (let k = 0; k < colsRepeated; k++) {
|
|
1038
|
+
// For ODS (spreadsheets), we skip empty cells to avoid massive ASTs (millions of cells)
|
|
1039
|
+
// but for ODP/ODT (presentation/text), cells are part of a defined table grid
|
|
1040
|
+
// Also include cells that have children (e.g., image nodes) even if no text
|
|
1041
|
+
if (cellText || children.length > 0 || fileType !== 'ods') {
|
|
1042
|
+
const cellNode = {
|
|
1043
|
+
type: 'cell',
|
|
1044
|
+
text: cellText,
|
|
1045
|
+
children: children,
|
|
1046
|
+
metadata: { row: rowIndex, col: colIndex }
|
|
1047
|
+
};
|
|
1048
|
+
if (config.includeRawContent) {
|
|
1049
|
+
cellNode.rawContent = cell.toString();
|
|
1050
|
+
}
|
|
1051
|
+
cells.push(cellNode);
|
|
1052
|
+
}
|
|
1053
|
+
colIndex++;
|
|
1054
|
+
}
|
|
1055
|
+
}
|
|
1056
|
+
// Add row(s)
|
|
1057
|
+
if (cells.length > 0) {
|
|
1058
|
+
for (let k = 0; k < rowsRepeated; k++) {
|
|
1059
|
+
const rowNode = {
|
|
1060
|
+
type: 'row',
|
|
1061
|
+
children: JSON.parse(JSON.stringify(cells)),
|
|
1062
|
+
metadata: undefined
|
|
1063
|
+
};
|
|
1064
|
+
// Fix row index in metadata for repeated rows
|
|
1065
|
+
if (k > 0) {
|
|
1066
|
+
rowNode.children?.forEach(c => {
|
|
1067
|
+
if (c.metadata && 'row' in c.metadata) {
|
|
1068
|
+
c.metadata.row = rowIndex;
|
|
1069
|
+
}
|
|
1070
|
+
});
|
|
1071
|
+
}
|
|
1072
|
+
if (config.includeRawContent) {
|
|
1073
|
+
rowNode.rawContent = row.toString();
|
|
1074
|
+
}
|
|
1075
|
+
rows.push(rowNode);
|
|
1076
|
+
rowIndex++;
|
|
1077
|
+
}
|
|
1078
|
+
}
|
|
1079
|
+
else {
|
|
1080
|
+
rowIndex += rowsRepeated;
|
|
1081
|
+
}
|
|
1082
|
+
}
|
|
1083
|
+
const sheetNode = {
|
|
1084
|
+
type: 'sheet',
|
|
1085
|
+
children: rows,
|
|
1086
|
+
metadata: { sheetName }
|
|
1087
|
+
};
|
|
1088
|
+
if (config.includeRawContent) {
|
|
1089
|
+
sheetNode.rawContent = table.toString();
|
|
1090
|
+
}
|
|
1091
|
+
content.push(sheetNode);
|
|
1092
|
+
}
|
|
1093
|
+
}
|
|
1094
|
+
}
|
|
1095
|
+
// ODP: Presentation
|
|
1096
|
+
else if (fileType === 'odp') {
|
|
1097
|
+
const presentation = (0, xmlUtils_1.getElementsByTagName)(body, "office:presentation")[0];
|
|
1098
|
+
if (presentation) {
|
|
1099
|
+
const pages = (0, xmlUtils_1.getDirectChildren)(presentation, "draw:page");
|
|
1100
|
+
const odpNotes = [];
|
|
1101
|
+
for (let i = 0; i < pages.length; i++) {
|
|
1102
|
+
const page = pages[i];
|
|
1103
|
+
const slideNode = {
|
|
1104
|
+
type: 'slide',
|
|
1105
|
+
children: [],
|
|
1106
|
+
metadata: { slideNumber: i + 1 }
|
|
1107
|
+
};
|
|
1108
|
+
// Separate page content and notes
|
|
1109
|
+
let noteNode = undefined;
|
|
1110
|
+
const pageChildren = page.childNodes;
|
|
1111
|
+
if (pageChildren) {
|
|
1112
|
+
for (let j = 0; j < pageChildren.length; j++) {
|
|
1113
|
+
const child = pageChildren[j];
|
|
1114
|
+
if (child.nodeType === 1) { // Element
|
|
1115
|
+
const element = child;
|
|
1116
|
+
if (element.tagName === "presentation:notes") {
|
|
1117
|
+
if (!config.ignoreNotes) {
|
|
1118
|
+
noteNode = {
|
|
1119
|
+
type: 'note',
|
|
1120
|
+
children: [],
|
|
1121
|
+
metadata: {
|
|
1122
|
+
slideNumber: i + 1,
|
|
1123
|
+
noteId: `slide-note-${i + 1}`
|
|
1124
|
+
}
|
|
1125
|
+
};
|
|
1126
|
+
traverse(element, noteNode.children);
|
|
1127
|
+
}
|
|
1128
|
+
continue;
|
|
1129
|
+
}
|
|
1130
|
+
traverse(element, slideNode.children);
|
|
1131
|
+
}
|
|
1132
|
+
}
|
|
1133
|
+
}
|
|
1134
|
+
if (config.includeRawContent) {
|
|
1135
|
+
slideNode.rawContent = page.toString();
|
|
1136
|
+
}
|
|
1137
|
+
content.push(slideNode);
|
|
1138
|
+
if (noteNode && noteNode.children && noteNode.children.length > 0) {
|
|
1139
|
+
if (config.putNotesAtLast) {
|
|
1140
|
+
odpNotes.push(noteNode);
|
|
1141
|
+
}
|
|
1142
|
+
else {
|
|
1143
|
+
content.push(noteNode);
|
|
1144
|
+
}
|
|
1145
|
+
}
|
|
1146
|
+
}
|
|
1147
|
+
if (odpNotes.length > 0) {
|
|
1148
|
+
content.push(...odpNotes);
|
|
1149
|
+
}
|
|
1150
|
+
}
|
|
1151
|
+
}
|
|
1152
|
+
// ODT: Text Document (and generic fallback)
|
|
1153
|
+
else {
|
|
1154
|
+
const textDoc = (0, xmlUtils_1.getElementsByTagName)(body, "office:text")[0];
|
|
1155
|
+
if (textDoc) {
|
|
1156
|
+
traverse(textDoc, content);
|
|
1157
|
+
}
|
|
1158
|
+
}
|
|
1159
|
+
};
|
|
1160
|
+
if (mainContentFile) {
|
|
1161
|
+
parseContentXml(mainContentFile.content.toString());
|
|
1162
|
+
}
|
|
1163
|
+
// Attachments
|
|
1164
|
+
const attachments = [];
|
|
1165
|
+
const mediaFiles = files.filter(f => f.path.match(/(Pictures|media)\/.*/));
|
|
1166
|
+
// ODP/ODT Chart Extraction
|
|
1167
|
+
if (config.extractAttachments) {
|
|
1168
|
+
const objectFiles = files.filter(f => f.path.match(/Object \d+\/content\.xml/));
|
|
1169
|
+
for (const objFile of objectFiles) {
|
|
1170
|
+
const objXml = (0, xmlUtils_1.parseXmlString)(objFile.content.toString());
|
|
1171
|
+
const isChart = (0, xmlUtils_1.getElementsByTagName)(objXml, "chart:chart").length > 0;
|
|
1172
|
+
if (isChart) {
|
|
1173
|
+
const objectId = objFile.path.split('/')[0];
|
|
1174
|
+
const attachment = {
|
|
1175
|
+
type: 'chart',
|
|
1176
|
+
mimeType: 'application/vnd.oasis.opendocument.chart',
|
|
1177
|
+
data: objFile.content.toString('base64'),
|
|
1178
|
+
name: objectId,
|
|
1179
|
+
extension: 'xml'
|
|
1180
|
+
};
|
|
1181
|
+
// Extract data from chart XML
|
|
1182
|
+
const chartData = (0, chartUtils_1.extractChartData)(objFile.content);
|
|
1183
|
+
if (chartData.rawTexts.length > 0) {
|
|
1184
|
+
attachment.chartData = chartData;
|
|
1185
|
+
}
|
|
1186
|
+
attachments.push(attachment);
|
|
1187
|
+
}
|
|
1188
|
+
}
|
|
1189
|
+
}
|
|
1190
|
+
if (config.extractAttachments) {
|
|
1191
|
+
for (const media of mediaFiles) {
|
|
1192
|
+
const attachment = (0, imageUtils_1.createAttachment)(media.path.split('/').pop() || 'image', media.content);
|
|
1193
|
+
attachments.push(attachment);
|
|
1194
|
+
if (config.ocr) {
|
|
1195
|
+
if (attachment.mimeType.startsWith('image/')) {
|
|
1196
|
+
try {
|
|
1197
|
+
attachment.ocrText = (await (0, ocrUtils_1.performOcr)(media.content, config.ocrLanguage)).trim();
|
|
1198
|
+
}
|
|
1199
|
+
catch (e) {
|
|
1200
|
+
(0, errorUtils_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
|
|
1201
|
+
}
|
|
1202
|
+
}
|
|
1203
|
+
}
|
|
1204
|
+
}
|
|
1205
|
+
}
|
|
1206
|
+
const metaFile = files.find(f => f.path.match(metaFileRegex));
|
|
1207
|
+
const metadata = metaFile ? (0, xmlUtils_1.parseOfficeMetadata)(metaFile.content.toString()) : {};
|
|
1208
|
+
// Helper: Resolve ODS chart cell references to actual values
|
|
1209
|
+
// ODS charts often link to cell ranges (e.g., [Sheet1.$A$1:.$A$5]) instead of embedding values
|
|
1210
|
+
const resolveChartReferences = (chartData, nodes) => {
|
|
1211
|
+
const getValuesFromReference = (ref) => {
|
|
1212
|
+
// Remove brackets: [Sheet.$A$1:.$A$5] -> Sheet.$A$1:.$A$5
|
|
1213
|
+
const cleanRef = ref.replace(/^\[|\]$/g, '');
|
|
1214
|
+
const [startPart, endPart] = cleanRef.split(':');
|
|
1215
|
+
const lastDotIdx = startPart.lastIndexOf('.');
|
|
1216
|
+
if (lastDotIdx === -1)
|
|
1217
|
+
return [ref];
|
|
1218
|
+
const sheetName = startPart.substring(0, lastDotIdx).replace(/^'|'$/g, '');
|
|
1219
|
+
const startCoord = startPart.substring(lastDotIdx + 1).replace(/\$/g, '');
|
|
1220
|
+
let endCoord = startCoord;
|
|
1221
|
+
if (endPart) {
|
|
1222
|
+
if (endPart.startsWith('.')) {
|
|
1223
|
+
endCoord = endPart.substring(1).replace(/\$/g, '');
|
|
1224
|
+
}
|
|
1225
|
+
else {
|
|
1226
|
+
const endLastDotIdx = endPart.lastIndexOf('.');
|
|
1227
|
+
endCoord = endPart.substring(endLastDotIdx + 1).replace(/\$/g, '');
|
|
1228
|
+
}
|
|
1229
|
+
}
|
|
1230
|
+
const parseCoord = (coord) => {
|
|
1231
|
+
const colMatch = coord.match(/[A-Z]+/);
|
|
1232
|
+
const rowMatch = coord.match(/\d+/);
|
|
1233
|
+
if (!colMatch || !rowMatch)
|
|
1234
|
+
return null;
|
|
1235
|
+
const colStr = colMatch[0];
|
|
1236
|
+
let colIdx = 0;
|
|
1237
|
+
for (let i = 0; i < colStr.length; i++) {
|
|
1238
|
+
colIdx = colIdx * 26 + (colStr.charCodeAt(i) - 'A'.charCodeAt(0) + 1);
|
|
1239
|
+
}
|
|
1240
|
+
colIdx -= 1;
|
|
1241
|
+
const rowIdx = parseInt(rowMatch[0]) - 1;
|
|
1242
|
+
return { r: rowIdx, c: colIdx };
|
|
1243
|
+
};
|
|
1244
|
+
const start = parseCoord(startCoord);
|
|
1245
|
+
const end = parseCoord(endCoord);
|
|
1246
|
+
if (!start || !end)
|
|
1247
|
+
return [ref];
|
|
1248
|
+
const sheet = nodes.find(n => n.type === 'sheet' && n.metadata?.sheetName === sheetName);
|
|
1249
|
+
if (!sheet || !sheet.children)
|
|
1250
|
+
return [ref];
|
|
1251
|
+
const values = [];
|
|
1252
|
+
// Collect all matching cells
|
|
1253
|
+
for (const row of sheet.children) {
|
|
1254
|
+
if (row.children) {
|
|
1255
|
+
for (const cell of row.children) {
|
|
1256
|
+
const meta = cell.metadata;
|
|
1257
|
+
if (meta && meta.row >= start.r && meta.row <= end.r && meta.col >= start.c && meta.col <= end.c) {
|
|
1258
|
+
values.push(cell.text || '');
|
|
1259
|
+
}
|
|
1260
|
+
}
|
|
1261
|
+
}
|
|
1262
|
+
}
|
|
1263
|
+
return values.length > 0 ? values : [];
|
|
1264
|
+
};
|
|
1265
|
+
// Resolve DataSets
|
|
1266
|
+
for (const ds of chartData.dataSets) {
|
|
1267
|
+
const newValues = [];
|
|
1268
|
+
for (const val of ds.values) {
|
|
1269
|
+
if (val.startsWith('['))
|
|
1270
|
+
newValues.push(...getValuesFromReference(val));
|
|
1271
|
+
else
|
|
1272
|
+
newValues.push(val);
|
|
1273
|
+
}
|
|
1274
|
+
ds.values = newValues;
|
|
1275
|
+
}
|
|
1276
|
+
// Resolve Labels
|
|
1277
|
+
const newLabels = [];
|
|
1278
|
+
for (const label of chartData.labels) {
|
|
1279
|
+
if (label.startsWith('['))
|
|
1280
|
+
newLabels.push(...getValuesFromReference(label));
|
|
1281
|
+
else
|
|
1282
|
+
newLabels.push(label);
|
|
1283
|
+
}
|
|
1284
|
+
chartData.labels = newLabels;
|
|
1285
|
+
// Rebuild rawTexts
|
|
1286
|
+
chartData.rawTexts = [];
|
|
1287
|
+
if (chartData.title)
|
|
1288
|
+
chartData.rawTexts.push(chartData.title);
|
|
1289
|
+
for (const ds of chartData.dataSets) {
|
|
1290
|
+
if (ds.name)
|
|
1291
|
+
chartData.rawTexts.push(ds.name);
|
|
1292
|
+
chartData.rawTexts.push(...chartData.labels);
|
|
1293
|
+
chartData.rawTexts.push(...ds.values);
|
|
1294
|
+
}
|
|
1295
|
+
};
|
|
1296
|
+
// Apply resolution to all chart attachments
|
|
1297
|
+
for (const att of attachments) {
|
|
1298
|
+
if (att.type === 'chart' && att.chartData) {
|
|
1299
|
+
resolveChartReferences(att.chartData, content);
|
|
1300
|
+
}
|
|
1301
|
+
}
|
|
1302
|
+
// Link OCR and Chart text to content nodes
|
|
1303
|
+
// Link OCR and Chart text to content nodes (with heuristic for unlinked images)
|
|
1304
|
+
const assignAttachmentData = (nodes) => {
|
|
1305
|
+
// Step 1: Identify unused image attachments globally
|
|
1306
|
+
const usedAttachmentNames = new Set();
|
|
1307
|
+
const traverseForNames = (ns) => {
|
|
1308
|
+
for (const n of ns) {
|
|
1309
|
+
if (n.metadata && 'attachmentName' in n.metadata) {
|
|
1310
|
+
const name = n.metadata.attachmentName;
|
|
1311
|
+
if (name)
|
|
1312
|
+
usedAttachmentNames.add(name);
|
|
1313
|
+
}
|
|
1314
|
+
if (n.children)
|
|
1315
|
+
traverseForNames(n.children);
|
|
1316
|
+
}
|
|
1317
|
+
};
|
|
1318
|
+
traverseForNames(nodes);
|
|
1319
|
+
const unusedImages = attachments.filter(a => a.type === 'image' && a.name && !usedAttachmentNames.has(a.name));
|
|
1320
|
+
let unusedImageIndex = 0;
|
|
1321
|
+
const processNode = (node) => {
|
|
1322
|
+
if ((node.type === 'image' || node.type === 'chart') && node.metadata && 'attachmentName' in node.metadata) {
|
|
1323
|
+
let attachmentName = node.metadata.attachmentName;
|
|
1324
|
+
// Heuristic: If name is empty, try to assign an unused image attachment
|
|
1325
|
+
if (!attachmentName && node.type === 'image' && unusedImageIndex < unusedImages.length) {
|
|
1326
|
+
const fallbackAtt = unusedImages[unusedImageIndex++];
|
|
1327
|
+
attachmentName = fallbackAtt.name;
|
|
1328
|
+
node.metadata.attachmentName = attachmentName;
|
|
1329
|
+
}
|
|
1330
|
+
if (attachmentName) {
|
|
1331
|
+
const attachment = attachments.find(a => a.name === attachmentName);
|
|
1332
|
+
if (attachment) {
|
|
1333
|
+
if (attachment.ocrText) {
|
|
1334
|
+
node.text = attachment.ocrText;
|
|
1335
|
+
}
|
|
1336
|
+
if (attachment.chartData && node.type === 'chart') {
|
|
1337
|
+
node.text = attachment.chartData.rawTexts.join(config.newlineDelimiter ?? '\n');
|
|
1338
|
+
}
|
|
1339
|
+
}
|
|
1340
|
+
}
|
|
1341
|
+
}
|
|
1342
|
+
// Internal recursion
|
|
1343
|
+
if (node.children) {
|
|
1344
|
+
node.children.forEach(processNode);
|
|
1345
|
+
}
|
|
1346
|
+
};
|
|
1347
|
+
nodes.forEach(processNode);
|
|
1348
|
+
};
|
|
1349
|
+
assignAttachmentData(content);
|
|
1350
|
+
// Create combined styleMap for metadata (matches DOCX format)
|
|
1351
|
+
const combinedStyleMap = {};
|
|
1352
|
+
for (const styleName in styleMap) {
|
|
1353
|
+
combinedStyleMap[styleName] = {
|
|
1354
|
+
formatting: styleMap[styleName],
|
|
1355
|
+
alignment: paragraphStyleMap[styleName]?.alignment
|
|
1356
|
+
};
|
|
1357
|
+
}
|
|
1358
|
+
// Also add styles that only have alignment
|
|
1359
|
+
for (const styleName in paragraphStyleMap) {
|
|
1360
|
+
if (!combinedStyleMap[styleName]) {
|
|
1361
|
+
combinedStyleMap[styleName] = {
|
|
1362
|
+
formatting: {},
|
|
1363
|
+
alignment: paragraphStyleMap[styleName]?.alignment
|
|
1364
|
+
};
|
|
1365
|
+
}
|
|
1366
|
+
}
|
|
1367
|
+
// Append notes to content if configured
|
|
1368
|
+
if (config.putNotesAtLast && notes.length > 0) {
|
|
1369
|
+
content.push(...notes);
|
|
1370
|
+
}
|
|
1371
|
+
return {
|
|
1372
|
+
type: fileType,
|
|
1373
|
+
metadata: {
|
|
1374
|
+
...metadata,
|
|
1375
|
+
styleMap: combinedStyleMap
|
|
1376
|
+
},
|
|
1377
|
+
content: content,
|
|
1378
|
+
attachments: attachments,
|
|
1379
|
+
toText: () => content.map(c => {
|
|
1380
|
+
const getText = (node) => {
|
|
1381
|
+
let t = '';
|
|
1382
|
+
if (node.children && node.children.length > 0) {
|
|
1383
|
+
// Check if children have their own children (container vs leaf)
|
|
1384
|
+
// If children are leaf nodes (text/image), join with empty string
|
|
1385
|
+
// If children are container nodes (paragraphs/rows), join with newline
|
|
1386
|
+
const hasGrandChildren = node.children.some(child => child.children && child.children.length > 0);
|
|
1387
|
+
const separator = hasGrandChildren ? (config.newlineDelimiter ?? '\n') : '';
|
|
1388
|
+
t += node.children.map(getText).filter(t => t != '').join(separator);
|
|
1389
|
+
}
|
|
1390
|
+
else {
|
|
1391
|
+
t += node.text || '';
|
|
1392
|
+
}
|
|
1393
|
+
return t;
|
|
1394
|
+
};
|
|
1395
|
+
return getText(c);
|
|
1396
|
+
}).filter(t => t != '').join(config.newlineDelimiter ?? '\n')
|
|
1397
|
+
};
|
|
1398
|
+
};
|
|
1399
|
+
exports.parseOpenOffice = parseOpenOffice;
|