officeparser 6.0.7 → 6.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +136 -52
- package/dist/OfficeParser.d.ts +10 -1
- package/dist/OfficeParser.js +44 -56
- package/dist/cli.d.ts +20 -0
- package/dist/cli.js +117 -0
- package/dist/index.d.ts +4 -4
- package/dist/index.js +7 -59
- package/dist/index.mjs +18 -0
- package/dist/officeparser.browser.d.ts +133 -3
- package/dist/officeparser.browser.iife.js +115 -0
- package/dist/officeparser.browser.mjs +114 -0
- package/dist/parsers/ExcelParser.d.ts +1 -1
- package/dist/parsers/ExcelParser.js +76 -68
- package/dist/parsers/OpenOfficeParser.d.ts +1 -1
- package/dist/parsers/OpenOfficeParser.js +224 -159
- package/dist/parsers/PdfParser.d.ts +1 -1
- package/dist/parsers/PdfParser.js +98 -94
- package/dist/parsers/PowerPointParser.d.ts +1 -1
- package/dist/parsers/PowerPointParser.js +188 -179
- package/dist/parsers/RtfParser.d.ts +21 -1
- package/dist/parsers/RtfParser.js +117 -48
- package/dist/parsers/WordParser.d.ts +2 -1
- package/dist/parsers/WordParser.js +214 -123
- package/dist/sbom.cdx.json +1807 -0
- package/dist/types.d.ts +123 -3
- package/dist/utils/chartUtils.js +2 -0
- package/dist/utils/dateUtils.d.ts +17 -0
- package/dist/utils/dateUtils.js +69 -0
- package/dist/utils/envUtils.d.ts +24 -0
- package/dist/utils/envUtils.js +69 -0
- package/dist/utils/moduleLoader.d.ts +2 -1
- package/dist/utils/moduleLoader.js +9 -39
- package/dist/utils/ocrUtils.d.ts +16 -12
- package/dist/utils/ocrUtils.js +186 -25
- package/dist/utils/xmlUtils.d.ts +80 -9
- package/dist/utils/xmlUtils.js +236 -18
- package/dist/utils/zipUtils.js +6 -47
- package/package.json +31 -16
- package/dist/officeParserBundle@6.0.7.js +0 -154
- package/dist/officeparser.browser.js +0 -154
|
@@ -23,12 +23,12 @@
|
|
|
23
23
|
*/
|
|
24
24
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
25
25
|
exports.parseOpenOffice = void 0;
|
|
26
|
-
const
|
|
27
|
-
const
|
|
28
|
-
const
|
|
29
|
-
const
|
|
30
|
-
const
|
|
31
|
-
const
|
|
26
|
+
const chartUtils_js_1 = require("../utils/chartUtils.js");
|
|
27
|
+
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
28
|
+
const imageUtils_js_1 = require("../utils/imageUtils.js");
|
|
29
|
+
const ocrUtils_js_1 = require("../utils/ocrUtils.js");
|
|
30
|
+
const xmlUtils_js_1 = require("../utils/xmlUtils.js");
|
|
31
|
+
const zipUtils_js_1 = require("../utils/zipUtils.js");
|
|
32
32
|
/**
|
|
33
33
|
* Parses an OpenOffice document (.odt, .odp, .ods) and extracts content.
|
|
34
34
|
*
|
|
@@ -43,7 +43,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
43
43
|
const metaFileRegex = /meta\.xml/;
|
|
44
44
|
const stylesFileRegex = /styles\.xml/;
|
|
45
45
|
const mimetypeFileRegex = /mimetype/;
|
|
46
|
-
const files = await (0,
|
|
46
|
+
const files = await (0, zipUtils_js_1.extractFiles)(buffer, x => !!x.match(contentFileRegex) ||
|
|
47
47
|
!!x.match(objectContentFileRegex) ||
|
|
48
48
|
!!x.match(metaFileRegex) ||
|
|
49
49
|
!!x.match(stylesFileRegex) ||
|
|
@@ -70,17 +70,22 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
70
70
|
const styleMap = {};
|
|
71
71
|
const paragraphStyleMap = {};
|
|
72
72
|
const listCounters = {}; // Track item index per listId/level
|
|
73
|
+
let currentListId = null;
|
|
74
|
+
let lastListType = null;
|
|
75
|
+
let lastListStyle = null;
|
|
76
|
+
let listIdCounter = 0;
|
|
77
|
+
let lastWasList = false;
|
|
73
78
|
// Helper to parse styles
|
|
74
79
|
const parseStyles = (xmlString) => {
|
|
75
|
-
const xml = (0,
|
|
76
|
-
const styles = (0,
|
|
80
|
+
const xml = (0, xmlUtils_js_1.parseXmlString)(xmlString);
|
|
81
|
+
const styles = (0, xmlUtils_js_1.getElementsByTagName)(xml, "style:style");
|
|
77
82
|
for (const style of styles) {
|
|
78
83
|
const name = style.getAttribute("style:name");
|
|
79
84
|
if (!name)
|
|
80
85
|
continue;
|
|
81
86
|
const styleInfo = {};
|
|
82
87
|
// Parse paragraph properties for alignment and drop caps
|
|
83
|
-
const paraProps = (0,
|
|
88
|
+
const paraProps = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "style:paragraph-properties");
|
|
84
89
|
if (paraProps) {
|
|
85
90
|
const textAlign = paraProps.getAttribute("fo:text-align");
|
|
86
91
|
if (textAlign) {
|
|
@@ -97,7 +102,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
97
102
|
}
|
|
98
103
|
}
|
|
99
104
|
// Detect Drop Caps
|
|
100
|
-
const dropCap = (0,
|
|
105
|
+
const dropCap = (0, xmlUtils_js_1.getFirstElementByTagName)(paraProps, "style:drop-cap");
|
|
101
106
|
if (dropCap) {
|
|
102
107
|
styleInfo.dropCap = true;
|
|
103
108
|
}
|
|
@@ -106,9 +111,9 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
106
111
|
paragraphStyleMap[name] = styleInfo;
|
|
107
112
|
}
|
|
108
113
|
// Parse text properties
|
|
109
|
-
const textProps = (0,
|
|
114
|
+
const textProps = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "style:text-properties");
|
|
110
115
|
// Parse table cell properties (for ODS background)
|
|
111
|
-
const cellProps = (0,
|
|
116
|
+
const cellProps = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "style:table-cell-properties");
|
|
112
117
|
const formatting = {};
|
|
113
118
|
if (cellProps) {
|
|
114
119
|
const bgColor = cellProps.getAttribute("fo:background-color");
|
|
@@ -170,7 +175,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
170
175
|
* @param linkMetadata - Metadata inherited from parent link
|
|
171
176
|
* @returns Object containing text and children
|
|
172
177
|
*/
|
|
173
|
-
const parseInlineContent = (node, styleMap, config, notes, paragraphStyleMap, parentFormatting = {}, linkMetadata) => {
|
|
178
|
+
const parseInlineContent = (node, styleMap, config, notes, paragraphStyleMap, parentFormatting = {}, linkMetadata, sourceXml = '') => {
|
|
174
179
|
const children = [];
|
|
175
180
|
let fullText = '';
|
|
176
181
|
if (!node.childNodes)
|
|
@@ -189,7 +194,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
189
194
|
});
|
|
190
195
|
}
|
|
191
196
|
}
|
|
192
|
-
else if (
|
|
197
|
+
else if ((0, xmlUtils_js_1.isElement)(child)) {
|
|
193
198
|
const element = child;
|
|
194
199
|
const tagName = element.tagName;
|
|
195
200
|
if (tagName === 'text:s') {
|
|
@@ -221,14 +226,14 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
221
226
|
type: 'text',
|
|
222
227
|
text: '\n',
|
|
223
228
|
formatting: parentFormatting,
|
|
224
|
-
metadata:
|
|
229
|
+
metadata: { ...(linkMetadata || {}), isLineBreak: true }
|
|
225
230
|
});
|
|
226
231
|
}
|
|
227
232
|
else if (tagName === 'text:span') {
|
|
228
233
|
// Formatted text span
|
|
229
234
|
const styleName = element.getAttribute("text:style-name");
|
|
230
235
|
const formatting = styleName ? { ...parentFormatting, ...styleMap[styleName] } : parentFormatting;
|
|
231
|
-
const spanContent = parseInlineContent(element, styleMap, config, notes, paragraphStyleMap, formatting, linkMetadata);
|
|
236
|
+
const spanContent = parseInlineContent(element, styleMap, config, notes, paragraphStyleMap, formatting, linkMetadata, sourceXml);
|
|
232
237
|
fullText += spanContent.text;
|
|
233
238
|
children.push(...spanContent.children);
|
|
234
239
|
}
|
|
@@ -237,7 +242,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
237
242
|
const href = element.getAttribute('xlink:href') || '';
|
|
238
243
|
const linkType = href.startsWith('#') ? 'internal' : 'external';
|
|
239
244
|
const newLinkMetadata = { link: href, linkType: linkType };
|
|
240
|
-
const linkContent = parseInlineContent(element, styleMap, config, notes, paragraphStyleMap, parentFormatting, newLinkMetadata);
|
|
245
|
+
const linkContent = parseInlineContent(element, styleMap, config, notes, paragraphStyleMap, parentFormatting, newLinkMetadata, sourceXml);
|
|
241
246
|
fullText += linkContent.text;
|
|
242
247
|
children.push(...linkContent.children);
|
|
243
248
|
}
|
|
@@ -245,14 +250,14 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
245
250
|
// Footnote or endnote
|
|
246
251
|
const noteClass = (element.getAttribute('text:note-class') || 'footnote');
|
|
247
252
|
const noteId = element.getAttribute('text:id') || element.getAttribute('xml:id') || undefined;
|
|
248
|
-
const noteBody = (0,
|
|
253
|
+
const noteBody = (0, xmlUtils_js_1.getFirstElementByTagName)(element, "text:note-body");
|
|
249
254
|
if (noteBody) {
|
|
250
255
|
// Extract note content recursively
|
|
251
|
-
const notePs = (0,
|
|
256
|
+
const notePs = (0, xmlUtils_js_1.getElementsByTagName)(noteBody, "text:p");
|
|
252
257
|
const noteChildren = [];
|
|
253
258
|
let noteText = '';
|
|
254
259
|
for (const np of notePs) {
|
|
255
|
-
const npContent = parseParagraphContent(np, paragraphStyleMap, styleMap, config);
|
|
260
|
+
const npContent = parseParagraphContent(np, paragraphStyleMap, styleMap, config, sourceXml);
|
|
256
261
|
noteText += (noteText ? ' ' : '') + npContent.text;
|
|
257
262
|
const npNode = {
|
|
258
263
|
type: 'paragraph',
|
|
@@ -284,8 +289,8 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
284
289
|
const frame = element;
|
|
285
290
|
// Extract alt text
|
|
286
291
|
let altText = '';
|
|
287
|
-
const svgTitle = (0,
|
|
288
|
-
const svgDesc = (0,
|
|
292
|
+
const svgTitle = (0, xmlUtils_js_1.getFirstElementByTagName)(frame, "svg:title");
|
|
293
|
+
const svgDesc = (0, xmlUtils_js_1.getFirstElementByTagName)(frame, "svg:desc");
|
|
289
294
|
if (svgTitle && svgTitle.textContent) {
|
|
290
295
|
altText = svgTitle.textContent;
|
|
291
296
|
}
|
|
@@ -294,7 +299,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
294
299
|
}
|
|
295
300
|
// Extract image href
|
|
296
301
|
let imageHref = '';
|
|
297
|
-
const drawImages = (0,
|
|
302
|
+
const drawImages = (0, xmlUtils_js_1.getElementsByTagName)(frame, "draw:image");
|
|
298
303
|
if (drawImages.length > 0) {
|
|
299
304
|
imageHref = drawImages[0].getAttribute("xlink:href") || '';
|
|
300
305
|
if (imageHref) {
|
|
@@ -312,7 +317,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
312
317
|
}
|
|
313
318
|
};
|
|
314
319
|
if (config.includeRawContent) {
|
|
315
|
-
imageNode.rawContent =
|
|
320
|
+
imageNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frame, sourceXml, config);
|
|
316
321
|
}
|
|
317
322
|
children.push(imageNode);
|
|
318
323
|
}
|
|
@@ -330,14 +335,14 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
330
335
|
* @param config - Parser configuration
|
|
331
336
|
* @returns Object containing text, children, alignment, and style info
|
|
332
337
|
*/
|
|
333
|
-
const parseParagraphContent = (node, paraStyleMap, styleMap, config) => {
|
|
338
|
+
const parseParagraphContent = (node, paraStyleMap, styleMap, config, sourceXml) => {
|
|
334
339
|
// Get paragraph style for alignment and drop caps
|
|
335
340
|
const paraStyle = node.getAttribute("text:style-name");
|
|
336
341
|
const styleInfo = paraStyle ? paraStyleMap[paraStyle] : undefined;
|
|
337
342
|
const alignment = styleInfo?.alignment;
|
|
338
343
|
const dropCap = styleInfo?.dropCap;
|
|
339
344
|
// Parse content recursively using the new helper
|
|
340
|
-
const content = parseInlineContent(node, styleMap, config, notes, paraStyleMap);
|
|
345
|
+
const content = parseInlineContent(node, styleMap, config, notes, paraStyleMap, {}, undefined, sourceXml);
|
|
341
346
|
// Add style name to metadata of children if they don't have one
|
|
342
347
|
if (paraStyle) {
|
|
343
348
|
content.children.forEach(child => {
|
|
@@ -391,6 +396,31 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
391
396
|
}
|
|
392
397
|
return { text: content.text, children: content.children, alignment, style: paraStyle || undefined };
|
|
393
398
|
};
|
|
399
|
+
/**
|
|
400
|
+
* Splits paragraph content into multiple segments based on line breaks.
|
|
401
|
+
* Used to handle soft line breaks within list items.
|
|
402
|
+
*
|
|
403
|
+
* @param pContent - The content of a single paragraph
|
|
404
|
+
* @returns Array of content segments
|
|
405
|
+
*/
|
|
406
|
+
const splitParagraphByBreaks = (pContent) => {
|
|
407
|
+
const segments = [];
|
|
408
|
+
let currentText = "";
|
|
409
|
+
let currentChildren = [];
|
|
410
|
+
for (const child of pContent.children) {
|
|
411
|
+
if (child.type === "text" && child.metadata?.isLineBreak) {
|
|
412
|
+
segments.push({ text: currentText, children: currentChildren });
|
|
413
|
+
currentText = "";
|
|
414
|
+
currentChildren = [];
|
|
415
|
+
}
|
|
416
|
+
else {
|
|
417
|
+
currentText += child.text || "";
|
|
418
|
+
currentChildren.push(child);
|
|
419
|
+
}
|
|
420
|
+
}
|
|
421
|
+
segments.push({ text: currentText, children: currentChildren });
|
|
422
|
+
return segments;
|
|
423
|
+
};
|
|
394
424
|
/**
|
|
395
425
|
* Helper to parse a table node and extract its structure.
|
|
396
426
|
* Properly creates table → row → cell hierarchy with metadata.
|
|
@@ -401,15 +431,15 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
401
431
|
* @param config - Parser configuration
|
|
402
432
|
* @returns Table content node with proper structure
|
|
403
433
|
*/
|
|
404
|
-
const parseTable = (tableNode, paraStyleMap, styleMap, config) => {
|
|
434
|
+
const parseTable = (tableNode, paraStyleMap, styleMap, config, sourceXml) => {
|
|
405
435
|
const rows = [];
|
|
406
436
|
// Use getDirectChildren to avoid nested table rows
|
|
407
|
-
const tableRows = (0,
|
|
437
|
+
const tableRows = (0, xmlUtils_js_1.getDirectChildren)(tableNode, "table:table-row");
|
|
408
438
|
let rowIndex = 0;
|
|
409
439
|
for (const row of tableRows) {
|
|
410
440
|
const cells = [];
|
|
411
441
|
// Use getDirectChildren to avoid nested table cells
|
|
412
|
-
const tableCells = (0,
|
|
442
|
+
const tableCells = (0, xmlUtils_js_1.getDirectChildren)(row, "table:table-cell");
|
|
413
443
|
const rowsRepeated = parseInt(row.getAttribute("table:number-rows-repeated") || "1");
|
|
414
444
|
let colIndex = 0;
|
|
415
445
|
for (const cell of tableCells) {
|
|
@@ -424,10 +454,10 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
424
454
|
return;
|
|
425
455
|
for (let i = 0; i < node.childNodes.length; i++) {
|
|
426
456
|
const child = node.childNodes[i];
|
|
427
|
-
if (
|
|
457
|
+
if ((0, xmlUtils_js_1.isElement)(child)) { // Element
|
|
428
458
|
const element = child;
|
|
429
459
|
if (element.tagName === "text:p" || element.tagName === "text:h") {
|
|
430
|
-
const pContent = parseParagraphContent(element, paraStyleMap, styleMap, config);
|
|
460
|
+
const pContent = parseParagraphContent(element, paraStyleMap, styleMap, config, sourceXml);
|
|
431
461
|
const pNode = {
|
|
432
462
|
type: element.tagName === "text:h" ? 'heading' : 'paragraph',
|
|
433
463
|
text: pContent.text,
|
|
@@ -446,7 +476,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
446
476
|
pNode.metadata.level = parseInt(element.getAttribute("text:outline-level") || "1");
|
|
447
477
|
}
|
|
448
478
|
if (config.includeRawContent) {
|
|
449
|
-
pNode.rawContent =
|
|
479
|
+
pNode.rawContent = (0, xmlUtils_js_1.getRawContent)(element, sourceXml, config);
|
|
450
480
|
}
|
|
451
481
|
cellChildren.push(pNode);
|
|
452
482
|
cellTextRef.value += pContent.text;
|
|
@@ -457,7 +487,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
457
487
|
}
|
|
458
488
|
else if (element.tagName === "table:table") {
|
|
459
489
|
// Recursive call for nested table
|
|
460
|
-
const nestedTableNode = parseTable(element, paraStyleMap, styleMap, config);
|
|
490
|
+
const nestedTableNode = parseTable(element, paraStyleMap, styleMap, config, sourceXml);
|
|
461
491
|
cellChildren.push(nestedTableNode);
|
|
462
492
|
}
|
|
463
493
|
else if (element.tagName === "draw:frame" || element.tagName === "draw:text-box") {
|
|
@@ -487,7 +517,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
487
517
|
if (rowSpan > 1)
|
|
488
518
|
cellMetadata.rowSpan = rowSpan;
|
|
489
519
|
if (config.includeRawContent) {
|
|
490
|
-
cellNode.rawContent =
|
|
520
|
+
cellNode.rawContent = (0, xmlUtils_js_1.getRawContent)(cell, sourceXml, config);
|
|
491
521
|
}
|
|
492
522
|
cells.push(cellNode);
|
|
493
523
|
colIndex++;
|
|
@@ -508,7 +538,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
508
538
|
});
|
|
509
539
|
}
|
|
510
540
|
if (config.includeRawContent) {
|
|
511
|
-
rowNode.rawContent =
|
|
541
|
+
rowNode.rawContent = (0, xmlUtils_js_1.getRawContent)(row, sourceXml, config);
|
|
512
542
|
}
|
|
513
543
|
rows.push(rowNode);
|
|
514
544
|
rowIndex++;
|
|
@@ -520,18 +550,20 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
520
550
|
};
|
|
521
551
|
};
|
|
522
552
|
const parseContentXml = (xmlString) => {
|
|
523
|
-
const xml = (0,
|
|
524
|
-
const body = (0,
|
|
553
|
+
const xml = (0, xmlUtils_js_1.parseXmlString)(xmlString, { locator: config.includeRawContent });
|
|
554
|
+
const body = (0, xmlUtils_js_1.getFirstElementByTagName)(xml, "office:body");
|
|
555
|
+
if (!body)
|
|
556
|
+
return;
|
|
525
557
|
// Parse automatic styles (local to content.xml)
|
|
526
|
-
const automaticStyles = (0,
|
|
558
|
+
const automaticStyles = (0, xmlUtils_js_1.getFirstElementByTagName)(xml, "office:automatic-styles");
|
|
527
559
|
if (automaticStyles) {
|
|
528
|
-
const styles = (0,
|
|
560
|
+
const styles = (0, xmlUtils_js_1.getElementsByTagName)(automaticStyles, "style:style");
|
|
529
561
|
for (const style of styles) {
|
|
530
562
|
const name = style.getAttribute("style:name");
|
|
531
563
|
if (!name)
|
|
532
564
|
continue;
|
|
533
565
|
// Parse paragraph properties for alignment
|
|
534
|
-
const paraProps = (0,
|
|
566
|
+
const paraProps = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "style:paragraph-properties");
|
|
535
567
|
const styleInfo = {};
|
|
536
568
|
if (paraProps) {
|
|
537
569
|
const textAlign = paraProps.getAttribute("fo:text-align");
|
|
@@ -548,14 +580,14 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
548
580
|
styleInfo.alignment = alignMap[textAlign];
|
|
549
581
|
}
|
|
550
582
|
}
|
|
551
|
-
const dropCap = (0,
|
|
583
|
+
const dropCap = (0, xmlUtils_js_1.getFirstElementByTagName)(paraProps, "style:drop-cap");
|
|
552
584
|
if (dropCap)
|
|
553
585
|
styleInfo.dropCap = true;
|
|
554
586
|
}
|
|
555
587
|
if (Object.keys(styleInfo).length > 0) {
|
|
556
588
|
paragraphStyleMap[name] = styleInfo;
|
|
557
589
|
}
|
|
558
|
-
const textProps = (0,
|
|
590
|
+
const textProps = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "style:text-properties");
|
|
559
591
|
if (textProps) {
|
|
560
592
|
const formatting = {};
|
|
561
593
|
if (textProps.getAttribute("fo:font-weight") === "bold" || textProps.getAttribute("style:font-weight-asian") === "bold")
|
|
@@ -593,6 +625,19 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
593
625
|
}
|
|
594
626
|
}
|
|
595
627
|
}
|
|
628
|
+
// Start traversal
|
|
629
|
+
const officeBody = (0, xmlUtils_js_1.getFirstElementByTagName)(xml, "office:body");
|
|
630
|
+
if (officeBody) {
|
|
631
|
+
const bodyContent = (0, xmlUtils_js_1.getDirectChildren)(officeBody, "office:text")[0] ||
|
|
632
|
+
(0, xmlUtils_js_1.getDirectChildren)(officeBody, "office:presentation")[0] ||
|
|
633
|
+
(0, xmlUtils_js_1.getDirectChildren)(officeBody, "office:spreadsheet")[0];
|
|
634
|
+
if (bodyContent) {
|
|
635
|
+
const bodyChildren = (0, xmlUtils_js_1.getDirectChildren)(bodyContent, "*");
|
|
636
|
+
for (const child of bodyChildren) {
|
|
637
|
+
traverse(child, content, false, xmlString);
|
|
638
|
+
}
|
|
639
|
+
}
|
|
640
|
+
}
|
|
596
641
|
/**
|
|
597
642
|
* Recursively traverses a node and its children to extract content.
|
|
598
643
|
* Properly handles paragraphs, headings, tables, lists, and frames.
|
|
@@ -600,10 +645,11 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
600
645
|
* @param node - The element to traverse
|
|
601
646
|
* @param targetArray - The array to push extracted content nodes to
|
|
602
647
|
* @param forceHeading - If true, treats all paragraphs as headings (used for slide titles)
|
|
648
|
+
* @param sourceXml - The source XML string for raw content extraction
|
|
603
649
|
*/
|
|
604
|
-
|
|
650
|
+
function traverse(node, targetArray, forceHeading = false, sourceXml) {
|
|
605
651
|
if (node.tagName === "text:p") {
|
|
606
|
-
const pContent = parseParagraphContent(node, paragraphStyleMap, styleMap, config);
|
|
652
|
+
const pContent = parseParagraphContent(node, paragraphStyleMap, styleMap, config, sourceXml);
|
|
607
653
|
const type = (forceHeading || (node.getAttribute("text:style-name") || '').toLowerCase().includes('title')) ? 'heading' : 'paragraph';
|
|
608
654
|
const pNode = {
|
|
609
655
|
type,
|
|
@@ -621,13 +667,14 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
621
667
|
if (Object.keys(pNode.metadata || {}).length === 0)
|
|
622
668
|
delete pNode.metadata;
|
|
623
669
|
if (config.includeRawContent) {
|
|
624
|
-
pNode.rawContent =
|
|
670
|
+
pNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
|
|
625
671
|
}
|
|
626
672
|
targetArray.push(pNode);
|
|
673
|
+
lastWasList = false;
|
|
627
674
|
}
|
|
628
675
|
else if (node.tagName === "text:h") {
|
|
629
676
|
const level = parseInt(node.getAttribute("text:outline-level") || "1");
|
|
630
|
-
const hContent = parseParagraphContent(node, paragraphStyleMap, styleMap, config);
|
|
677
|
+
const hContent = parseParagraphContent(node, paragraphStyleMap, styleMap, config, sourceXml);
|
|
631
678
|
const hNode = {
|
|
632
679
|
type: 'heading',
|
|
633
680
|
text: hContent.text,
|
|
@@ -639,27 +686,27 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
639
686
|
}
|
|
640
687
|
};
|
|
641
688
|
if (config.includeRawContent) {
|
|
642
|
-
hNode.rawContent =
|
|
689
|
+
hNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
|
|
643
690
|
}
|
|
644
691
|
targetArray.push(hNode);
|
|
692
|
+
lastWasList = false;
|
|
645
693
|
}
|
|
646
694
|
else if (node.tagName === "table:table") {
|
|
647
695
|
// Parse table with proper structure
|
|
648
|
-
const tableNode = parseTable(node, paragraphStyleMap, styleMap, config);
|
|
696
|
+
const tableNode = parseTable(node, paragraphStyleMap, styleMap, config, sourceXml);
|
|
649
697
|
if (config.includeRawContent) {
|
|
650
|
-
tableNode.rawContent =
|
|
698
|
+
tableNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
|
|
651
699
|
}
|
|
652
700
|
targetArray.push(tableNode);
|
|
701
|
+
lastWasList = false;
|
|
653
702
|
}
|
|
654
703
|
else if (node.tagName === "text:list") {
|
|
655
704
|
// Parse list structure with proper listId tracking
|
|
656
|
-
const listItems = (0,
|
|
657
|
-
// Get list style name to use as listId (or generate one)
|
|
658
|
-
const listStyleName = node.getAttribute("text:style-name") || node.getAttribute("xml:id");
|
|
659
|
-
const listId = listStyleName || `list-${targetArray.length}`;
|
|
705
|
+
const listItems = (0, xmlUtils_js_1.getDirectChildren)(node, "text:list-item");
|
|
660
706
|
// Determine list type by checking the list style definition
|
|
661
707
|
let listType = 'unordered';
|
|
662
708
|
let isVisible = false;
|
|
709
|
+
const listStyleName = node.getAttribute("text:style-name") || node.getAttribute("xml:id");
|
|
663
710
|
let styleNameToCheck = listStyleName;
|
|
664
711
|
// If no style name, check parent list for inherited style
|
|
665
712
|
if (!styleNameToCheck) {
|
|
@@ -675,15 +722,15 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
675
722
|
}
|
|
676
723
|
// Try to find list style in automatic styles to determine type and visibility
|
|
677
724
|
if (styleNameToCheck) {
|
|
678
|
-
const automaticStyles = (0,
|
|
725
|
+
const automaticStyles = (0, xmlUtils_js_1.getFirstElementByTagName)((0, xmlUtils_js_1.parseXmlString)(mainContentFile?.content.toString() || ''), "office:automatic-styles");
|
|
679
726
|
if (automaticStyles) {
|
|
680
|
-
const listStyles = (0,
|
|
727
|
+
const listStyles = (0, xmlUtils_js_1.getElementsByTagName)(automaticStyles, "text:list-style");
|
|
681
728
|
for (const listStyle of listStyles) {
|
|
682
729
|
if (listStyle.getAttribute("style:name") === styleNameToCheck) {
|
|
683
730
|
// Check if it has bullet or number level styles
|
|
684
|
-
const bulletLevels = (0,
|
|
685
|
-
const numberLevels = (0,
|
|
686
|
-
const imageLevels = (0,
|
|
731
|
+
const bulletLevels = (0, xmlUtils_js_1.getElementsByTagName)(listStyle, "text:list-level-style-bullet");
|
|
732
|
+
const numberLevels = (0, xmlUtils_js_1.getElementsByTagName)(listStyle, "text:list-level-style-number");
|
|
733
|
+
const imageLevels = (0, xmlUtils_js_1.getElementsByTagName)(listStyle, "text:list-level-style-image");
|
|
687
734
|
if (numberLevels.length > 0) {
|
|
688
735
|
listType = 'ordered';
|
|
689
736
|
isVisible = numberLevels.some(l => !!l.getAttribute("style:num-format"));
|
|
@@ -700,13 +747,13 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
700
747
|
}
|
|
701
748
|
// Also check in styles.xml if still unordered and hidden
|
|
702
749
|
if (stylesFile && !isVisible) {
|
|
703
|
-
const stylesXml = (0,
|
|
704
|
-
const listStyles = (0,
|
|
750
|
+
const stylesXml = (0, xmlUtils_js_1.parseXmlString)(stylesFile.content.toString());
|
|
751
|
+
const listStyles = (0, xmlUtils_js_1.getElementsByTagName)(stylesXml, "text:list-style");
|
|
705
752
|
for (const listStyle of listStyles) {
|
|
706
753
|
if (listStyle.getAttribute("style:name") === styleNameToCheck) {
|
|
707
|
-
const bulletLevels = (0,
|
|
708
|
-
const numberLevels = (0,
|
|
709
|
-
const imageLevels = (0,
|
|
754
|
+
const bulletLevels = (0, xmlUtils_js_1.getElementsByTagName)(listStyle, "text:list-level-style-bullet");
|
|
755
|
+
const numberLevels = (0, xmlUtils_js_1.getElementsByTagName)(listStyle, "text:list-level-style-number");
|
|
756
|
+
const imageLevels = (0, xmlUtils_js_1.getElementsByTagName)(listStyle, "text:list-level-style-image");
|
|
710
757
|
if (numberLevels.length > 0) {
|
|
711
758
|
listType = 'ordered';
|
|
712
759
|
isVisible = numberLevels.some(l => !!l.getAttribute("style:num-format"));
|
|
@@ -725,19 +772,38 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
725
772
|
// If the list is not visible, it's likely a layout list used by Impress.
|
|
726
773
|
// We should traverse its items and treat their content as regular nodes.
|
|
727
774
|
if (!isVisible) {
|
|
775
|
+
lastWasList = false;
|
|
728
776
|
for (let i = 0; i < listItems.length; i++) {
|
|
729
777
|
const item = listItems[i];
|
|
730
778
|
if (item.childNodes) {
|
|
731
779
|
for (let j = 0; j < item.childNodes.length; j++) {
|
|
732
780
|
const child = item.childNodes[j];
|
|
733
|
-
if (
|
|
734
|
-
traverse(child, targetArray, forceHeading);
|
|
781
|
+
if ((0, xmlUtils_js_1.isElement)(child)) { // Element
|
|
782
|
+
traverse(child, targetArray, forceHeading, sourceXml);
|
|
735
783
|
}
|
|
736
784
|
}
|
|
737
785
|
}
|
|
738
786
|
}
|
|
739
787
|
return;
|
|
740
788
|
}
|
|
789
|
+
// List Continuity Logic:
|
|
790
|
+
// If this list follows another list of the same type and style, or we are in ODP and it's sequential,
|
|
791
|
+
// we should reuse the previous listId to maintain numbering.
|
|
792
|
+
const isODP = fileType === 'odp';
|
|
793
|
+
const sameStyle = styleNameToCheck && styleNameToCheck === lastListStyle;
|
|
794
|
+
const sameType = listType === lastListType;
|
|
795
|
+
let listId;
|
|
796
|
+
if (lastWasList && (sameStyle || (isODP && sameType))) {
|
|
797
|
+
listId = currentListId;
|
|
798
|
+
}
|
|
799
|
+
else {
|
|
800
|
+
// New list
|
|
801
|
+
listId = styleNameToCheck || `list-${++listIdCounter}`;
|
|
802
|
+
currentListId = listId;
|
|
803
|
+
lastListType = listType;
|
|
804
|
+
lastListStyle = styleNameToCheck;
|
|
805
|
+
}
|
|
806
|
+
lastWasList = true;
|
|
741
807
|
// Calculate indentation level by counting parent text:list elements
|
|
742
808
|
let indentation = 0;
|
|
743
809
|
let parent = node.parentNode;
|
|
@@ -758,63 +824,61 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
758
824
|
// Process each list item
|
|
759
825
|
for (let i = 0; i < listItems.length; i++) {
|
|
760
826
|
const item = listItems[i];
|
|
761
|
-
|
|
762
|
-
listCounters[listId][indentKey]++;
|
|
763
|
-
const itemIndex = listCounters[listId][indentKey];
|
|
764
|
-
// Reset deeper levels when we encounter an item at this level
|
|
765
|
-
for (let k = indentation + 1; k < 10; k++) {
|
|
766
|
-
if (listCounters[listId][k.toString()] !== undefined) {
|
|
767
|
-
listCounters[listId][k.toString()] = -1;
|
|
768
|
-
}
|
|
769
|
-
}
|
|
827
|
+
let hasIndexedThisItem = false;
|
|
770
828
|
// Iterate over direct children of list item (paragraphs, headings, nested lists)
|
|
771
829
|
if (item.childNodes) {
|
|
772
830
|
for (let j = 0; j < item.childNodes.length; j++) {
|
|
773
831
|
const child = item.childNodes[j];
|
|
774
|
-
if (
|
|
832
|
+
if ((0, xmlUtils_js_1.isElement)(child)) { // Element
|
|
775
833
|
const element = child;
|
|
776
|
-
if (element.tagName === "text:p") {
|
|
777
|
-
|
|
778
|
-
|
|
779
|
-
|
|
780
|
-
|
|
781
|
-
|
|
782
|
-
|
|
783
|
-
|
|
784
|
-
indentation,
|
|
785
|
-
itemIndex,
|
|
786
|
-
listId,
|
|
787
|
-
alignment: pContent.alignment || 'left',
|
|
788
|
-
style: pContent.style
|
|
834
|
+
if (element.tagName === "text:p" || element.tagName === "text:h") {
|
|
835
|
+
if (!hasIndexedThisItem) {
|
|
836
|
+
listCounters[listId][indentKey]++;
|
|
837
|
+
hasIndexedThisItem = true;
|
|
838
|
+
for (let k = indentation + 1; k < 10; k++) {
|
|
839
|
+
if (listCounters[listId][k.toString()] !== undefined) {
|
|
840
|
+
listCounters[listId][k.toString()] = -1;
|
|
841
|
+
}
|
|
789
842
|
}
|
|
790
|
-
}
|
|
791
|
-
|
|
792
|
-
|
|
793
|
-
|
|
794
|
-
|
|
795
|
-
|
|
796
|
-
|
|
797
|
-
|
|
798
|
-
|
|
799
|
-
|
|
800
|
-
|
|
801
|
-
|
|
802
|
-
|
|
803
|
-
|
|
804
|
-
|
|
805
|
-
|
|
806
|
-
|
|
807
|
-
|
|
808
|
-
|
|
843
|
+
}
|
|
844
|
+
const itemIndex = listCounters[listId][indentKey];
|
|
845
|
+
const pContent = parseParagraphContent(element, paragraphStyleMap, styleMap, config, sourceXml);
|
|
846
|
+
const segments = splitParagraphByBreaks(pContent);
|
|
847
|
+
for (let k = 0; k < segments.length; k++) {
|
|
848
|
+
const segment = segments[k];
|
|
849
|
+
if (!segment.text.trim() && segment.children.length === 0)
|
|
850
|
+
continue;
|
|
851
|
+
const isFirst = k === 0;
|
|
852
|
+
const nodeType = isFirst ? 'list' : 'paragraph';
|
|
853
|
+
const node = {
|
|
854
|
+
type: nodeType,
|
|
855
|
+
text: segment.text,
|
|
856
|
+
children: segment.children,
|
|
857
|
+
metadata: isFirst ? {
|
|
858
|
+
listType,
|
|
859
|
+
indentation,
|
|
860
|
+
itemIndex,
|
|
861
|
+
listId,
|
|
862
|
+
alignment: pContent.alignment || 'left',
|
|
863
|
+
style: pContent.style
|
|
864
|
+
} : {
|
|
865
|
+
alignment: pContent.alignment || 'left',
|
|
866
|
+
style: pContent.style
|
|
867
|
+
}
|
|
868
|
+
};
|
|
869
|
+
// Special case for headings in lists
|
|
870
|
+
if (isFirst && element.tagName === "text:h") {
|
|
871
|
+
const level = parseInt(element.getAttribute("text:outline-level") || "1");
|
|
872
|
+
node.metadata.level = level;
|
|
809
873
|
}
|
|
810
|
-
|
|
811
|
-
|
|
812
|
-
|
|
813
|
-
|
|
874
|
+
if (config.includeRawContent)
|
|
875
|
+
node.rawContent = (0, xmlUtils_js_1.getRawContent)(element, sourceXml, config);
|
|
876
|
+
targetArray.push(node);
|
|
877
|
+
}
|
|
814
878
|
}
|
|
815
879
|
else if (element.tagName === "text:list") {
|
|
816
880
|
// Recursive call for nested list
|
|
817
|
-
traverse(element, targetArray, forceHeading);
|
|
881
|
+
traverse(element, targetArray, forceHeading, sourceXml);
|
|
818
882
|
}
|
|
819
883
|
}
|
|
820
884
|
}
|
|
@@ -825,24 +889,24 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
825
889
|
const presClass = node.getAttribute("presentation:class");
|
|
826
890
|
const isHeading = presClass === "title" || presClass === "sub-title";
|
|
827
891
|
// In presentations, frames often contain text-boxes, images, tables, or objects
|
|
828
|
-
const textBox = (0,
|
|
829
|
-
const image = (0,
|
|
830
|
-
const table = (0,
|
|
831
|
-
const object = (0,
|
|
892
|
+
const textBox = (0, xmlUtils_js_1.getFirstElementByTagName)(node, "draw:text-box");
|
|
893
|
+
const image = (0, xmlUtils_js_1.getFirstElementByTagName)(node, "draw:image");
|
|
894
|
+
const table = (0, xmlUtils_js_1.getFirstElementByTagName)(node, "table:table");
|
|
895
|
+
const object = (0, xmlUtils_js_1.getFirstElementByTagName)(node, "draw:object");
|
|
832
896
|
if (textBox) {
|
|
833
|
-
traverse(textBox, targetArray, isHeading || forceHeading);
|
|
897
|
+
traverse(textBox, targetArray, isHeading || forceHeading, sourceXml);
|
|
834
898
|
}
|
|
835
899
|
else if (table) {
|
|
836
|
-
const tableNode = parseTable(table, paragraphStyleMap, styleMap, config);
|
|
900
|
+
const tableNode = parseTable(table, paragraphStyleMap, styleMap, config, sourceXml);
|
|
837
901
|
if (config.includeRawContent)
|
|
838
|
-
tableNode.rawContent =
|
|
902
|
+
tableNode.rawContent = (0, xmlUtils_js_1.getRawContent)(table, sourceXml, config);
|
|
839
903
|
targetArray.push(tableNode);
|
|
840
904
|
}
|
|
841
905
|
else if (image) {
|
|
842
906
|
// Extract alt text from svg:title or svg:desc
|
|
843
907
|
let altText = '';
|
|
844
|
-
const svgTitle = (0,
|
|
845
|
-
const svgDesc = (0,
|
|
908
|
+
const svgTitle = (0, xmlUtils_js_1.getFirstElementByTagName)(node, "svg:title");
|
|
909
|
+
const svgDesc = (0, xmlUtils_js_1.getFirstElementByTagName)(node, "svg:desc");
|
|
846
910
|
if (svgTitle && svgTitle.textContent) {
|
|
847
911
|
altText = svgTitle.textContent;
|
|
848
912
|
}
|
|
@@ -865,7 +929,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
865
929
|
}
|
|
866
930
|
};
|
|
867
931
|
if (config.includeRawContent) {
|
|
868
|
-
imageNode.rawContent =
|
|
932
|
+
imageNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
|
|
869
933
|
}
|
|
870
934
|
targetArray.push(imageNode);
|
|
871
935
|
}
|
|
@@ -877,7 +941,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
877
941
|
const objectPath = `${attachmentName}/content.xml`;
|
|
878
942
|
const objectFile = files.find(f => f.path === objectPath || f.path.endsWith(objectPath));
|
|
879
943
|
if (objectFile) {
|
|
880
|
-
const chartData = (0,
|
|
944
|
+
const chartData = (0, chartUtils_js_1.extractChartData)(objectFile.content);
|
|
881
945
|
const chartNode = {
|
|
882
946
|
type: 'chart',
|
|
883
947
|
text: chartData.rawTexts.join(" "),
|
|
@@ -887,7 +951,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
887
951
|
}
|
|
888
952
|
};
|
|
889
953
|
if (config.includeRawContent)
|
|
890
|
-
chartNode.rawContent =
|
|
954
|
+
chartNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
|
|
891
955
|
targetArray.push(chartNode);
|
|
892
956
|
}
|
|
893
957
|
else {
|
|
@@ -897,7 +961,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
897
961
|
metadata: { attachmentName: attachmentName }
|
|
898
962
|
};
|
|
899
963
|
if (config.includeRawContent)
|
|
900
|
-
chartNode.rawContent =
|
|
964
|
+
chartNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
|
|
901
965
|
targetArray.push(chartNode);
|
|
902
966
|
}
|
|
903
967
|
}
|
|
@@ -907,28 +971,29 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
907
971
|
if (node.childNodes) {
|
|
908
972
|
for (let i = 0; i < node.childNodes.length; i++) {
|
|
909
973
|
const child = node.childNodes[i];
|
|
910
|
-
if (
|
|
911
|
-
traverse(child, targetArray, forceHeading);
|
|
974
|
+
if ((0, xmlUtils_js_1.isElement)(child)) { // Element
|
|
975
|
+
traverse(child, targetArray, forceHeading, sourceXml);
|
|
912
976
|
}
|
|
913
977
|
}
|
|
914
978
|
}
|
|
915
979
|
}
|
|
916
|
-
}
|
|
980
|
+
}
|
|
981
|
+
;
|
|
917
982
|
// ODS: Spreadsheet
|
|
918
983
|
if (fileType === 'ods') {
|
|
919
|
-
const spreadsheet = (0,
|
|
984
|
+
const spreadsheet = (0, xmlUtils_js_1.getFirstElementByTagName)(body, "office:spreadsheet");
|
|
920
985
|
if (spreadsheet) {
|
|
921
|
-
const tables = (0,
|
|
986
|
+
const tables = (0, xmlUtils_js_1.getElementsByTagName)(spreadsheet, "table:table");
|
|
922
987
|
for (let i = 0; i < tables.length; i++) {
|
|
923
988
|
const table = tables[i];
|
|
924
989
|
const sheetName = table.getAttribute("table:name") || `Sheet${i + 1}`;
|
|
925
990
|
const rows = [];
|
|
926
|
-
const tableRows = (0,
|
|
991
|
+
const tableRows = (0, xmlUtils_js_1.getElementsByTagName)(table, "table:table-row");
|
|
927
992
|
let rowIndex = 0;
|
|
928
993
|
for (let r = 0; r < tableRows.length; r++) {
|
|
929
994
|
const row = tableRows[r];
|
|
930
995
|
const cells = [];
|
|
931
|
-
const tableCells = (0,
|
|
996
|
+
const tableCells = (0, xmlUtils_js_1.getElementsByTagName)(row, "table:table-cell");
|
|
932
997
|
let colIndex = 0;
|
|
933
998
|
const rowsRepeated = parseInt(row.getAttribute("table:number-rows-repeated") || "1");
|
|
934
999
|
for (let c = 0; c < tableCells.length; c++) {
|
|
@@ -937,11 +1002,11 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
937
1002
|
// Extract text from cell (paragraphs inside cell)
|
|
938
1003
|
let cellText = "";
|
|
939
1004
|
const children = [];
|
|
940
|
-
const ps = (0,
|
|
1005
|
+
const ps = (0, xmlUtils_js_1.getElementsByTagName)(cell, "text:p");
|
|
941
1006
|
for (let p = 0; p < ps.length; p++) {
|
|
942
1007
|
const para = ps[p];
|
|
943
1008
|
// Parse text:span elements for formatted text
|
|
944
|
-
const spans = (0,
|
|
1009
|
+
const spans = (0, xmlUtils_js_1.getElementsByTagName)(para, "text:span");
|
|
945
1010
|
if (spans.length > 0) {
|
|
946
1011
|
for (const span of spans) {
|
|
947
1012
|
const styleName = span.getAttribute("text:style-name");
|
|
@@ -973,12 +1038,12 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
973
1038
|
cellText += "\n";
|
|
974
1039
|
}
|
|
975
1040
|
// Check for embedded draw:frame (images) in cell
|
|
976
|
-
const drawFrames = (0,
|
|
1041
|
+
const drawFrames = (0, xmlUtils_js_1.getElementsByTagName)(cell, "draw:frame");
|
|
977
1042
|
for (const frame of drawFrames) {
|
|
978
1043
|
// Extract alt text from svg:title or svg:desc
|
|
979
1044
|
let altText = '';
|
|
980
|
-
const svgTitle = (0,
|
|
981
|
-
const svgDesc = (0,
|
|
1045
|
+
const svgTitle = (0, xmlUtils_js_1.getFirstElementByTagName)(frame, "svg:title");
|
|
1046
|
+
const svgDesc = (0, xmlUtils_js_1.getFirstElementByTagName)(frame, "svg:desc");
|
|
982
1047
|
if (svgTitle && svgTitle.textContent) {
|
|
983
1048
|
altText = svgTitle.textContent;
|
|
984
1049
|
}
|
|
@@ -987,7 +1052,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
987
1052
|
}
|
|
988
1053
|
// Extract image href
|
|
989
1054
|
let imageHref = '';
|
|
990
|
-
const drawImages = (0,
|
|
1055
|
+
const drawImages = (0, xmlUtils_js_1.getElementsByTagName)(frame, "draw:image");
|
|
991
1056
|
if (drawImages.length > 0) {
|
|
992
1057
|
const rawHref = drawImages[0].getAttribute("xlink:href");
|
|
993
1058
|
if (rawHref) {
|
|
@@ -997,7 +1062,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
997
1062
|
}
|
|
998
1063
|
// Extract chart object href
|
|
999
1064
|
let chartHref = '';
|
|
1000
|
-
const drawObjects = (0,
|
|
1065
|
+
const drawObjects = (0, xmlUtils_js_1.getElementsByTagName)(frame, "draw:object");
|
|
1001
1066
|
if (drawObjects.length > 0) {
|
|
1002
1067
|
const href = drawObjects[0].getAttribute("xlink:href");
|
|
1003
1068
|
if (href) {
|
|
@@ -1017,7 +1082,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1017
1082
|
}
|
|
1018
1083
|
};
|
|
1019
1084
|
if (config.includeRawContent) {
|
|
1020
|
-
imageNode.rawContent =
|
|
1085
|
+
imageNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frame, xmlString, config);
|
|
1021
1086
|
}
|
|
1022
1087
|
children.push(imageNode);
|
|
1023
1088
|
}
|
|
@@ -1046,7 +1111,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1046
1111
|
metadata: { row: rowIndex, col: colIndex }
|
|
1047
1112
|
};
|
|
1048
1113
|
if (config.includeRawContent) {
|
|
1049
|
-
cellNode.rawContent =
|
|
1114
|
+
cellNode.rawContent = (0, xmlUtils_js_1.getRawContent)(cell, xmlString, config);
|
|
1050
1115
|
}
|
|
1051
1116
|
cells.push(cellNode);
|
|
1052
1117
|
}
|
|
@@ -1070,7 +1135,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1070
1135
|
});
|
|
1071
1136
|
}
|
|
1072
1137
|
if (config.includeRawContent) {
|
|
1073
|
-
rowNode.rawContent =
|
|
1138
|
+
rowNode.rawContent = (0, xmlUtils_js_1.getRawContent)(row, xmlString, config);
|
|
1074
1139
|
}
|
|
1075
1140
|
rows.push(rowNode);
|
|
1076
1141
|
rowIndex++;
|
|
@@ -1086,7 +1151,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1086
1151
|
metadata: { sheetName }
|
|
1087
1152
|
};
|
|
1088
1153
|
if (config.includeRawContent) {
|
|
1089
|
-
sheetNode.rawContent =
|
|
1154
|
+
sheetNode.rawContent = (0, xmlUtils_js_1.getRawContent)(table, xmlString, config);
|
|
1090
1155
|
}
|
|
1091
1156
|
content.push(sheetNode);
|
|
1092
1157
|
}
|
|
@@ -1094,9 +1159,9 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1094
1159
|
}
|
|
1095
1160
|
// ODP: Presentation
|
|
1096
1161
|
else if (fileType === 'odp') {
|
|
1097
|
-
const presentation = (0,
|
|
1162
|
+
const presentation = (0, xmlUtils_js_1.getFirstElementByTagName)(body, "office:presentation");
|
|
1098
1163
|
if (presentation) {
|
|
1099
|
-
const pages = (0,
|
|
1164
|
+
const pages = (0, xmlUtils_js_1.getDirectChildren)(presentation, "draw:page");
|
|
1100
1165
|
const odpNotes = [];
|
|
1101
1166
|
for (let i = 0; i < pages.length; i++) {
|
|
1102
1167
|
const page = pages[i];
|
|
@@ -1111,7 +1176,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1111
1176
|
if (pageChildren) {
|
|
1112
1177
|
for (let j = 0; j < pageChildren.length; j++) {
|
|
1113
1178
|
const child = pageChildren[j];
|
|
1114
|
-
if (
|
|
1179
|
+
if ((0, xmlUtils_js_1.isElement)(child)) { // Element
|
|
1115
1180
|
const element = child;
|
|
1116
1181
|
if (element.tagName === "presentation:notes") {
|
|
1117
1182
|
if (!config.ignoreNotes) {
|
|
@@ -1123,16 +1188,16 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1123
1188
|
noteId: `slide-note-${i + 1}`
|
|
1124
1189
|
}
|
|
1125
1190
|
};
|
|
1126
|
-
traverse(element, noteNode.children);
|
|
1191
|
+
traverse(element, noteNode.children, false, xmlString);
|
|
1127
1192
|
}
|
|
1128
1193
|
continue;
|
|
1129
1194
|
}
|
|
1130
|
-
traverse(element, slideNode.children);
|
|
1195
|
+
traverse(element, slideNode.children, false, xmlString);
|
|
1131
1196
|
}
|
|
1132
1197
|
}
|
|
1133
1198
|
}
|
|
1134
1199
|
if (config.includeRawContent) {
|
|
1135
|
-
slideNode.rawContent =
|
|
1200
|
+
slideNode.rawContent = (0, xmlUtils_js_1.getRawContent)(page, xmlString, config);
|
|
1136
1201
|
}
|
|
1137
1202
|
content.push(slideNode);
|
|
1138
1203
|
if (noteNode && noteNode.children && noteNode.children.length > 0) {
|
|
@@ -1151,9 +1216,9 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1151
1216
|
}
|
|
1152
1217
|
// ODT: Text Document (and generic fallback)
|
|
1153
1218
|
else {
|
|
1154
|
-
const textDoc = (0,
|
|
1219
|
+
const textDoc = (0, xmlUtils_js_1.getFirstElementByTagName)(body, "office:text");
|
|
1155
1220
|
if (textDoc) {
|
|
1156
|
-
traverse(textDoc, content);
|
|
1221
|
+
traverse(textDoc, content, false, xmlString);
|
|
1157
1222
|
}
|
|
1158
1223
|
}
|
|
1159
1224
|
};
|
|
@@ -1167,8 +1232,8 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1167
1232
|
if (config.extractAttachments) {
|
|
1168
1233
|
const objectFiles = files.filter(f => f.path.match(/Object \d+\/content\.xml/));
|
|
1169
1234
|
for (const objFile of objectFiles) {
|
|
1170
|
-
const objXml = (0,
|
|
1171
|
-
const isChart = (0,
|
|
1235
|
+
const objXml = (0, xmlUtils_js_1.parseXmlString)(objFile.content.toString());
|
|
1236
|
+
const isChart = (0, xmlUtils_js_1.getElementsByTagName)(objXml, "chart:chart").length > 0;
|
|
1172
1237
|
if (isChart) {
|
|
1173
1238
|
const objectId = objFile.path.split('/')[0];
|
|
1174
1239
|
const attachment = {
|
|
@@ -1179,7 +1244,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1179
1244
|
extension: 'xml'
|
|
1180
1245
|
};
|
|
1181
1246
|
// Extract data from chart XML
|
|
1182
|
-
const chartData = (0,
|
|
1247
|
+
const chartData = (0, chartUtils_js_1.extractChartData)(objFile.content);
|
|
1183
1248
|
if (chartData.rawTexts.length > 0) {
|
|
1184
1249
|
attachment.chartData = chartData;
|
|
1185
1250
|
}
|
|
@@ -1189,22 +1254,22 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1189
1254
|
}
|
|
1190
1255
|
if (config.extractAttachments) {
|
|
1191
1256
|
for (const media of mediaFiles) {
|
|
1192
|
-
const attachment = (0,
|
|
1257
|
+
const attachment = (0, imageUtils_js_1.createAttachment)(media.path.split('/').pop() || 'image', media.content);
|
|
1193
1258
|
attachments.push(attachment);
|
|
1194
1259
|
if (config.ocr) {
|
|
1195
1260
|
if (attachment.mimeType.startsWith('image/')) {
|
|
1196
1261
|
try {
|
|
1197
|
-
attachment.ocrText = (await (0,
|
|
1262
|
+
attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(media.content, { language: config.ocrLanguage, ...config.ocrConfig })).trim();
|
|
1198
1263
|
}
|
|
1199
1264
|
catch (e) {
|
|
1200
|
-
(0,
|
|
1265
|
+
(0, errorUtils_js_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
|
|
1201
1266
|
}
|
|
1202
1267
|
}
|
|
1203
1268
|
}
|
|
1204
1269
|
}
|
|
1205
1270
|
}
|
|
1206
1271
|
const metaFile = files.find(f => f.path.match(metaFileRegex));
|
|
1207
|
-
const metadata = metaFile ? (0,
|
|
1272
|
+
const metadata = metaFile ? (0, xmlUtils_js_1.parseOfficeMetadata)(metaFile.content.toString()) : {};
|
|
1208
1273
|
// Helper: Resolve ODS chart cell references to actual values
|
|
1209
1274
|
// ODS charts often link to cell ranges (e.g., [Sheet1.$A$1:.$A$5]) instead of embedding values
|
|
1210
1275
|
const resolveChartReferences = (chartData, nodes) => {
|