officeparser 7.0.3 → 7.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +152 -18
- package/dist/OfficeGenerator.d.ts +1 -1
- package/dist/OfficeGenerator.js +16 -7
- package/dist/OfficeParser.js +6 -0
- package/dist/cli.d.ts +4 -0
- package/dist/cli.js +12 -3
- package/dist/defaults.js +27 -1
- package/dist/generators/BaseGenerator.d.ts +3 -3
- package/dist/generators/ChunkingGenerator.js +31 -4
- package/dist/generators/CsvGenerator.d.ts +1 -1
- package/dist/generators/HtmlGenerator.d.ts +2 -1
- package/dist/generators/HtmlGenerator.js +462 -40
- package/dist/generators/MarkdownGenerator.d.ts +1 -1
- package/dist/generators/MarkdownGenerator.js +3 -1
- package/dist/generators/PdfGenerator.d.ts +1 -1
- package/dist/generators/PdfGenerator.js +51 -10
- package/dist/generators/RtfGenerator.d.ts +2 -1
- package/dist/generators/RtfGenerator.js +43 -6
- package/dist/generators/TextGenerator.d.ts +1 -1
- package/dist/officeparser.browser.d.ts +377 -53
- package/dist/officeparser.browser.iife.js +380 -93
- package/dist/officeparser.browser.mjs +380 -93
- package/dist/parsers/CsvParser.js +6 -1
- package/dist/parsers/ExcelParser.js +69 -21
- package/dist/parsers/HtmlParser.js +15 -1
- package/dist/parsers/MarkdownParser.js +18 -10
- package/dist/parsers/OpenOfficeParser.js +61 -34
- package/dist/parsers/PdfParser.js +26 -1
- package/dist/parsers/PowerPointParser.js +168 -40
- package/dist/parsers/RtfParser.js +30 -24
- package/dist/parsers/WordParser.js +158 -11
- package/dist/sbom.cdx.json +100 -100
- package/dist/types.d.ts +383 -53
- package/dist/types.js +4 -0
- package/dist/utils/astUtils.d.ts +2 -2
- package/dist/utils/astUtils.js +2 -1
- package/dist/utils/configUtils.d.ts +5 -0
- package/dist/utils/configUtils.js +69 -2
- package/dist/utils/errorUtils.d.ts +20 -0
- package/dist/utils/errorUtils.js +39 -3
- package/dist/utils/moduleLoader.js +3 -3
- package/dist/utils/ocrUtils.js +271 -66
- package/dist/utils/xmlUtils.d.ts +17 -0
- package/dist/utils/xmlUtils.js +85 -1
- package/package.json +3 -2
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
3
|
exports.parseCsv = void 0;
|
|
4
4
|
const astUtils_js_1 = require("../utils/astUtils.js");
|
|
5
|
+
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
5
6
|
/**
|
|
6
7
|
* Parses a CSV file and extracts a single sheet with rows and cells.
|
|
7
8
|
*
|
|
@@ -10,6 +11,10 @@ const astUtils_js_1 = require("../utils/astUtils.js");
|
|
|
10
11
|
* @returns A promise resolving to the parsed AST
|
|
11
12
|
*/
|
|
12
13
|
const parseCsv = async (buffer, config) => {
|
|
14
|
+
// Honour cancellation requests before the character-by-character parsing loop starts.
|
|
15
|
+
// CSV has no OCR or async I/O, but very large files can still occupy the thread for a
|
|
16
|
+
// noticeable duration, so short-circuiting on an aborted signal is still worthwhile.
|
|
17
|
+
(0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
|
|
13
18
|
const textStr = buffer.toString('utf-8');
|
|
14
19
|
const delimiter = config.csvDelimiter;
|
|
15
20
|
const records = [];
|
|
@@ -105,6 +110,6 @@ const parseCsv = async (buffer, config) => {
|
|
|
105
110
|
.join(config.newlineDelimiter)
|
|
106
111
|
.replace(/\n{3,}/g, '\n\n');
|
|
107
112
|
};
|
|
108
|
-
return (0, astUtils_js_1.createAST)('csv', { title: 'Sheet1' }, [sheetNode], [], config, toTextSync);
|
|
113
|
+
return (0, astUtils_js_1.createAST)('csv', { title: 'Sheet1' }, [sheetNode], [], config, undefined, toTextSync);
|
|
109
114
|
};
|
|
110
115
|
exports.parseCsv = parseCsv;
|
|
@@ -40,6 +40,10 @@ const zipUtils_js_1 = require("../utils/zipUtils.js");
|
|
|
40
40
|
* @returns A promise resolving to the parsed AST
|
|
41
41
|
*/
|
|
42
42
|
const parseExcel = async (buffer, config) => {
|
|
43
|
+
// Honour cancellation requests immediately — before extracting the ZIP archive.
|
|
44
|
+
// XLSX parsing involves decompressing multiple XML sheets and potentially running OCR
|
|
45
|
+
// on embedded chart images, so short-circuiting here saves significant work.
|
|
46
|
+
(0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
|
|
43
47
|
const sheetsRegex = /xl\/worksheets\/sheet\d+.xml/g;
|
|
44
48
|
const drawingsRegex = /xl\/drawings\/drawing\d+.xml/g;
|
|
45
49
|
const chartsRegex = /xl\/charts\/chart\d+.xml/g;
|
|
@@ -47,18 +51,23 @@ const parseExcel = async (buffer, config) => {
|
|
|
47
51
|
const mediaFileRegex = /xl\/media\/.*/;
|
|
48
52
|
const corePropsFileRegex = /docProps\/core\.xml/;
|
|
49
53
|
const customPropsFileRegex = /docProps\/custom\.xml/;
|
|
54
|
+
const appPropsFileRegex = /docProps\/app\.xml/;
|
|
50
55
|
const relsRegex = /xl\/worksheets\/_rels\/sheet\d+\.xml\.rels/g;
|
|
51
56
|
const drawingRelsRegex = /xl\/drawings\/_rels\/drawing\d+\.xml\.rels/g;
|
|
57
|
+
const commentsRegex = /xl\/comments\d+\.xml/g;
|
|
52
58
|
const files = await (0, zipUtils_js_1.extractFiles)(buffer, (x) => !!x.match(sheetsRegex) ||
|
|
53
59
|
!!x.match(drawingsRegex) ||
|
|
54
60
|
!!x.match(chartsRegex) ||
|
|
61
|
+
(!config.ignoreComments && !!x.match(commentsRegex)) ||
|
|
55
62
|
x === stringsFilePath ||
|
|
56
63
|
x === 'xl/styles.xml' ||
|
|
57
64
|
x === 'xl/workbook.xml' ||
|
|
58
65
|
x === 'xl/_rels/workbook.xml.rels' ||
|
|
59
66
|
!!x.match(corePropsFileRegex) ||
|
|
60
67
|
!!x.match(customPropsFileRegex) ||
|
|
61
|
-
|
|
68
|
+
!!x.match(appPropsFileRegex) ||
|
|
69
|
+
(!!config.extractAttachments && (!!x.match(mediaFileRegex) || !!x.match(drawingRelsRegex))) ||
|
|
70
|
+
((!!config.extractAttachments || !config.ignoreComments) && !!x.match(relsRegex)));
|
|
62
71
|
const sharedStringsFile = files.find(f => f.path === stringsFilePath);
|
|
63
72
|
// Updated to store structured content (rich text runs) or simple string
|
|
64
73
|
const sharedStrings = [];
|
|
@@ -411,6 +420,52 @@ const parseExcel = async (buffer, config) => {
|
|
|
411
420
|
if (config.includeRawContent) {
|
|
412
421
|
rawContents.push(file.content.toString());
|
|
413
422
|
}
|
|
423
|
+
const sheetFilename = file.path.split('/').pop() || '';
|
|
424
|
+
const relsFilename = `xl/worksheets/_rels/${sheetFilename}.rels`;
|
|
425
|
+
const relsFile = files.find(f => f.path === relsFilename);
|
|
426
|
+
const drawingMap = {}; // rId -> drawingPath
|
|
427
|
+
const sheetCommentsMap = {};
|
|
428
|
+
if (relsFile) {
|
|
429
|
+
const relsXml = (0, xmlUtils_js_1.parseXmlString)(relsFile.content.toString());
|
|
430
|
+
const relationships = (0, xmlUtils_js_1.getElementsByTagName)(relsXml, "Relationship");
|
|
431
|
+
for (const rel of relationships) {
|
|
432
|
+
const id = rel.getAttribute("Id");
|
|
433
|
+
const target = rel.getAttribute("Target");
|
|
434
|
+
const type = rel.getAttribute("Type");
|
|
435
|
+
if (id && target && type) {
|
|
436
|
+
if (config.extractAttachments && type.includes('drawing')) {
|
|
437
|
+
drawingMap[id] = 'xl/drawings/' + target.replace('../drawings/', '');
|
|
438
|
+
}
|
|
439
|
+
else if (!config.ignoreComments && type.includes('comments')) {
|
|
440
|
+
const commentsPath = 'xl/' + target.replace('../', '');
|
|
441
|
+
const cFile = files.find(f => f.path === commentsPath);
|
|
442
|
+
if (cFile) {
|
|
443
|
+
const cXml = (0, xmlUtils_js_1.parseXmlString)(cFile.content.toString());
|
|
444
|
+
const commentNodes = (0, xmlUtils_js_1.getElementsByTagName)(cXml, "comment");
|
|
445
|
+
const authorsList = (0, xmlUtils_js_1.getElementsByTagName)(cXml, "author");
|
|
446
|
+
const authors = authorsList.map(a => a.textContent || '');
|
|
447
|
+
for (const cNode of commentNodes) {
|
|
448
|
+
const ref = cNode.getAttribute("ref");
|
|
449
|
+
const authorId = cNode.getAttribute("authorId");
|
|
450
|
+
const author = authorId !== null ? authors[parseInt(authorId)] : undefined;
|
|
451
|
+
const tNodes = (0, xmlUtils_js_1.getElementsByTagName)(cNode, "t");
|
|
452
|
+
const text = tNodes.map(t => t.textContent || '').join('');
|
|
453
|
+
if (ref && text) {
|
|
454
|
+
if (!sheetCommentsMap[ref])
|
|
455
|
+
sheetCommentsMap[ref] = [];
|
|
456
|
+
sheetCommentsMap[ref].push({
|
|
457
|
+
type: 'comment',
|
|
458
|
+
text: text,
|
|
459
|
+
children: [{ type: 'text', text: text, formatting: {} }],
|
|
460
|
+
metadata: author ? { author } : undefined
|
|
461
|
+
});
|
|
462
|
+
}
|
|
463
|
+
}
|
|
464
|
+
}
|
|
465
|
+
}
|
|
466
|
+
}
|
|
467
|
+
}
|
|
468
|
+
}
|
|
414
469
|
const rows = [];
|
|
415
470
|
const sheetXml = file.content.toString();
|
|
416
471
|
// regex to match <row> elements, capturing:
|
|
@@ -456,7 +511,7 @@ const parseExcel = async (buffer, config) => {
|
|
|
456
511
|
const typeMatch = cAttrs.match(/t="([a-zA-Z]+)"/);
|
|
457
512
|
const type = typeMatch ? typeMatch[1] : 'n'; // n = number (default)
|
|
458
513
|
const vMatch = cContent.match(/<v>([\s\S]*?)<\/v>/);
|
|
459
|
-
const tMatch = cContent.match(/<t
|
|
514
|
+
const tMatch = cContent.match(/<t\b[^>]*>([\s\S]*?)<\/t>/);
|
|
460
515
|
let text = '';
|
|
461
516
|
let cellNodes = [];
|
|
462
517
|
if (type === 's' && vMatch) {
|
|
@@ -473,7 +528,7 @@ const parseExcel = async (buffer, config) => {
|
|
|
473
528
|
}
|
|
474
529
|
}
|
|
475
530
|
else if (type === 'inlineStr' && tMatch) {
|
|
476
|
-
text = tMatch[1].trim();
|
|
531
|
+
text = (0, xmlUtils_js_1.decodeXmlEntities)(tMatch[1].trim());
|
|
477
532
|
}
|
|
478
533
|
else if (vMatch) {
|
|
479
534
|
text = vMatch[1].trim();
|
|
@@ -481,7 +536,9 @@ const parseExcel = async (buffer, config) => {
|
|
|
481
536
|
// Parse cell coordinate
|
|
482
537
|
const coordMatch = cAttrs.match(/r="([A-Z]+)(\d+)"/);
|
|
483
538
|
let colIndex;
|
|
539
|
+
let ref;
|
|
484
540
|
if (coordMatch) {
|
|
541
|
+
ref = coordMatch[1] + coordMatch[2];
|
|
485
542
|
colIndex = colToNumber(coordMatch[1]);
|
|
486
543
|
// If row index is missing in cell coord (unlikely but possible), use rowIndex
|
|
487
544
|
}
|
|
@@ -521,10 +578,12 @@ const parseExcel = async (buffer, config) => {
|
|
|
521
578
|
formatting: cellFormatting
|
|
522
579
|
});
|
|
523
580
|
}
|
|
581
|
+
const commentsNodeList = (ref && sheetCommentsMap[ref]) ? sheetCommentsMap[ref] : undefined;
|
|
524
582
|
const cellNode = {
|
|
525
583
|
type: 'cell',
|
|
526
584
|
text: text,
|
|
527
585
|
children: cellNodes,
|
|
586
|
+
comments: commentsNodeList,
|
|
528
587
|
metadata: { row: rowIndex, col: colIndex }
|
|
529
588
|
};
|
|
530
589
|
if (config.includeRawContent) {
|
|
@@ -547,23 +606,6 @@ const parseExcel = async (buffer, config) => {
|
|
|
547
606
|
}
|
|
548
607
|
// Handle Drawings in Sheet (images and charts)
|
|
549
608
|
if (config.extractAttachments) {
|
|
550
|
-
// Parse Sheet Rels to map drawing rIds
|
|
551
|
-
const sheetFilename = file.path.split('/').pop() || '';
|
|
552
|
-
const relsFilename = `xl/worksheets/_rels/${sheetFilename}.rels`;
|
|
553
|
-
const relsFile = files.find(f => f.path === relsFilename);
|
|
554
|
-
const drawingMap = {}; // rId -> drawingPath
|
|
555
|
-
if (relsFile) {
|
|
556
|
-
const relsXml = (0, xmlUtils_js_1.parseXmlString)(relsFile.content.toString());
|
|
557
|
-
const relationships = (0, xmlUtils_js_1.getElementsByTagName)(relsXml, "Relationship");
|
|
558
|
-
for (const rel of relationships) {
|
|
559
|
-
const id = rel.getAttribute("Id");
|
|
560
|
-
const target = rel.getAttribute("Target");
|
|
561
|
-
const type = rel.getAttribute("Type");
|
|
562
|
-
if (id && target && type && type.includes('drawing')) {
|
|
563
|
-
drawingMap[id] = 'xl/drawings/' + target.replace('../drawings/', '');
|
|
564
|
-
}
|
|
565
|
-
}
|
|
566
|
-
}
|
|
567
609
|
const drawingMatches = file.content.toString().match(/<drawing r:id="(.*?)"/g);
|
|
568
610
|
if (drawingMatches) {
|
|
569
611
|
for (const match of drawingMatches) {
|
|
@@ -633,6 +675,12 @@ const parseExcel = async (buffer, config) => {
|
|
|
633
675
|
if (Object.keys(customProperties).length > 0)
|
|
634
676
|
metadata.customProperties = customProperties;
|
|
635
677
|
}
|
|
678
|
+
const appPropsFile = files.find(f => f.path.match(appPropsFileRegex));
|
|
679
|
+
if (appPropsFile) {
|
|
680
|
+
const appProperties = (0, xmlUtils_js_1.parseOOXMLAppProperties)(appPropsFile.content.toString());
|
|
681
|
+
if (Object.keys(appProperties).length > 0)
|
|
682
|
+
metadata.nativeProperties = appProperties;
|
|
683
|
+
}
|
|
636
684
|
// Link OCR text and chart data to content nodes (like PPTX parser)
|
|
637
685
|
const assignAttachmentData = (nodes) => {
|
|
638
686
|
for (const node of nodes) {
|
|
@@ -677,6 +725,6 @@ const parseExcel = async (buffer, config) => {
|
|
|
677
725
|
};
|
|
678
726
|
return getText(c);
|
|
679
727
|
}).filter(t => t != '').join(config.newlineDelimiter);
|
|
680
|
-
return (0, astUtils_js_1.createAST)('xlsx', metadata, content, attachments, config, toTextSync);
|
|
728
|
+
return (0, astUtils_js_1.createAST)('xlsx', metadata, content, attachments, config, undefined, toTextSync);
|
|
681
729
|
};
|
|
682
730
|
exports.parseExcel = parseExcel;
|
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
3
|
exports.parseHtml = void 0;
|
|
4
4
|
const astUtils_js_1 = require("../utils/astUtils.js");
|
|
5
|
+
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
5
6
|
const parseAttributes = (attrString) => {
|
|
6
7
|
const attrs = {};
|
|
7
8
|
const regex = /([a-zA-Z0-9\-:]+)(?:\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s>]+)))?/g;
|
|
@@ -95,6 +96,10 @@ const parseHtmlTree = (html) => {
|
|
|
95
96
|
return root;
|
|
96
97
|
};
|
|
97
98
|
const parseHtml = async (buffer, config) => {
|
|
99
|
+
// Honour cancellation requests before the HTML tree is built and traversed.
|
|
100
|
+
// The custom recursive HTML parser can be expensive for large documents;
|
|
101
|
+
// rejecting early here prevents both the parsing and the subsequent AST construction.
|
|
102
|
+
(0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
|
|
98
103
|
const textStr = buffer.toString('utf-8');
|
|
99
104
|
const root = parseHtmlTree(textStr);
|
|
100
105
|
// Find head and body
|
|
@@ -122,6 +127,15 @@ const parseHtml = async (buffer, config) => {
|
|
|
122
127
|
if (titleNode && titleNode.children.length > 0 && titleNode.children[0].text) {
|
|
123
128
|
metadata.title = titleNode.children[0].text;
|
|
124
129
|
}
|
|
130
|
+
metadata.nativeProperties = {};
|
|
131
|
+
for (const child of head.children) {
|
|
132
|
+
if (child.tagName === 'meta') {
|
|
133
|
+
const name = child.attributes?.name || child.attributes?.property || child.attributes?.['http-equiv'];
|
|
134
|
+
if (name) {
|
|
135
|
+
metadata.nativeProperties[name] = child.attributes?.content || '';
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
}
|
|
125
139
|
const extractMeta = (name) => {
|
|
126
140
|
for (const child of head.children) {
|
|
127
141
|
if (child.tagName === 'meta' && (child.attributes?.name === name || child.attributes?.property === name)) {
|
|
@@ -534,6 +548,6 @@ const parseHtml = async (buffer, config) => {
|
|
|
534
548
|
return getText(n);
|
|
535
549
|
}).join(config.newlineDelimiter)
|
|
536
550
|
.replace(/\n{3,}/g, '\n\n'); // Normalize excessive whitespace
|
|
537
|
-
return (0, astUtils_js_1.createAST)('html', metadata, content, attachments, config, toTextSync);
|
|
551
|
+
return (0, astUtils_js_1.createAST)('html', metadata, content, attachments, config, undefined, toTextSync);
|
|
538
552
|
};
|
|
539
553
|
exports.parseHtml = parseHtml;
|
|
@@ -2,7 +2,12 @@
|
|
|
2
2
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
3
|
exports.parseMarkdown = void 0;
|
|
4
4
|
const astUtils_js_1 = require("../utils/astUtils.js");
|
|
5
|
+
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
5
6
|
const parseMarkdown = async (buffer, config) => {
|
|
7
|
+
// Honour cancellation requests before the line-by-line Markdown scanning loop begins.
|
|
8
|
+
// Markdown parsing is entirely synchronous and CPU-bound, so failing fast avoids
|
|
9
|
+
// processing content whose result will be discarded anyway.
|
|
10
|
+
(0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
|
|
6
11
|
let textStr = buffer.toString('utf-8');
|
|
7
12
|
textStr = textStr.replace(/\r\n/g, '\n');
|
|
8
13
|
const content = [];
|
|
@@ -16,11 +21,20 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
16
21
|
textStr = textStr.substring(endIdx + 5);
|
|
17
22
|
const lines = frontMatter.split('\n');
|
|
18
23
|
const customProps = {};
|
|
24
|
+
const nativeProps = {};
|
|
19
25
|
for (const line of lines) {
|
|
20
26
|
const match = line.match(/^([^:]+):\s*(.*)$/);
|
|
21
27
|
if (match) {
|
|
22
28
|
const key = match[1].trim();
|
|
23
29
|
let val = match[2].trim().replace(/^"(.*)"$/, '$1');
|
|
30
|
+
let parsedVal = val;
|
|
31
|
+
if (val === 'true')
|
|
32
|
+
parsedVal = true;
|
|
33
|
+
else if (val === 'false')
|
|
34
|
+
parsedVal = false;
|
|
35
|
+
else if (!isNaN(Number(val)) && val !== '')
|
|
36
|
+
parsedVal = Number(val);
|
|
37
|
+
nativeProps[key] = parsedVal;
|
|
24
38
|
if (key === 'title')
|
|
25
39
|
metadata.title = val;
|
|
26
40
|
else if (key === 'author')
|
|
@@ -32,20 +46,14 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
32
46
|
else if (key === 'description')
|
|
33
47
|
metadata.description = val;
|
|
34
48
|
else {
|
|
35
|
-
|
|
36
|
-
if (val === 'true')
|
|
37
|
-
customProps[key] = true;
|
|
38
|
-
else if (val === 'false')
|
|
39
|
-
customProps[key] = false;
|
|
40
|
-
else if (!isNaN(Number(val)) && val !== '')
|
|
41
|
-
customProps[key] = Number(val);
|
|
42
|
-
else
|
|
43
|
-
customProps[key] = val;
|
|
49
|
+
customProps[key] = parsedVal;
|
|
44
50
|
}
|
|
45
51
|
}
|
|
46
52
|
}
|
|
47
53
|
if (Object.keys(customProps).length > 0)
|
|
48
54
|
metadata.customProperties = customProps;
|
|
55
|
+
if (Object.keys(nativeProps).length > 0)
|
|
56
|
+
metadata.nativeProperties = nativeProps;
|
|
49
57
|
}
|
|
50
58
|
}
|
|
51
59
|
// Extract code blocks first to protect their contents
|
|
@@ -355,6 +363,6 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
355
363
|
return getText(n);
|
|
356
364
|
}).join(config.newlineDelimiter)
|
|
357
365
|
.replace(/\n{3,}/g, '\n\n'); // Normalize excessive whitespace
|
|
358
|
-
return (0, astUtils_js_1.createAST)('md', metadata, content, attachments, config, toTextSync);
|
|
366
|
+
return (0, astUtils_js_1.createAST)('md', metadata, content, attachments, config, undefined, toTextSync);
|
|
359
367
|
};
|
|
360
368
|
exports.parseMarkdown = parseMarkdown;
|
|
@@ -39,6 +39,10 @@ const zipUtils_js_1 = require("../utils/zipUtils.js");
|
|
|
39
39
|
* @returns A promise resolving to the parsed AST
|
|
40
40
|
*/
|
|
41
41
|
const parseOpenOffice = async (buffer, config) => {
|
|
42
|
+
// Honour cancellation requests immediately — before extracting the ZIP archive.
|
|
43
|
+
// ODF containers (ODT/ODS/ODP) bundle content.xml, styles.xml, and media files;
|
|
44
|
+
// aborting early avoids needlessly inflating and parsing all of those resources.
|
|
45
|
+
(0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
|
|
42
46
|
const contentFileRegex = /content\.xml/;
|
|
43
47
|
const objectContentFileRegex = /Object \d+\/content\.xml/;
|
|
44
48
|
const mediaFileRegex = /(Pictures|media)\/.*/;
|
|
@@ -153,9 +157,9 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
153
157
|
if (textPosition.startsWith("super"))
|
|
154
158
|
formatting.superscript = true;
|
|
155
159
|
}
|
|
156
|
-
if (Object.keys(formatting).length > 0)
|
|
157
|
-
styleMap[name] = formatting;
|
|
158
160
|
}
|
|
161
|
+
if (Object.keys(formatting).length > 0)
|
|
162
|
+
styleMap[name] = formatting;
|
|
159
163
|
}
|
|
160
164
|
};
|
|
161
165
|
if (stylesDom) {
|
|
@@ -306,11 +310,17 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
306
310
|
noteId: noteId
|
|
307
311
|
}
|
|
308
312
|
};
|
|
309
|
-
if (
|
|
310
|
-
|
|
313
|
+
if (children.length > 0 && children[children.length - 1].type === 'text') {
|
|
314
|
+
const precedingNode = children[children.length - 1];
|
|
315
|
+
if (!precedingNode.notes) {
|
|
316
|
+
precedingNode.notes = [];
|
|
317
|
+
}
|
|
318
|
+
precedingNode.notes.push(noteNode);
|
|
311
319
|
}
|
|
312
320
|
else {
|
|
313
|
-
|
|
321
|
+
const emptyTextNode = { type: 'text', text: '' };
|
|
322
|
+
emptyTextNode.notes = [noteNode];
|
|
323
|
+
children.push(emptyTextNode);
|
|
314
324
|
}
|
|
315
325
|
}
|
|
316
326
|
}
|
|
@@ -488,22 +498,33 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
488
498
|
const element = child;
|
|
489
499
|
if (element.tagName === "text:p" || element.tagName === "text:h") {
|
|
490
500
|
const pContent = parseParagraphContent(element, paraStyleMap, styleMap, config, sourceXml);
|
|
491
|
-
|
|
492
|
-
|
|
493
|
-
|
|
494
|
-
|
|
495
|
-
|
|
496
|
-
|
|
497
|
-
|
|
498
|
-
|
|
499
|
-
|
|
501
|
+
let pNode;
|
|
502
|
+
if (element.tagName === "text:h") {
|
|
503
|
+
pNode = {
|
|
504
|
+
type: 'heading',
|
|
505
|
+
text: pContent.text,
|
|
506
|
+
children: pContent.children,
|
|
507
|
+
metadata: {
|
|
508
|
+
level: parseInt(element.getAttribute("text:outline-level") || "1"),
|
|
509
|
+
...(pContent.alignment ? { alignment: pContent.alignment } : {}),
|
|
510
|
+
...(pContent.style ? { style: pContent.style } : {})
|
|
511
|
+
}
|
|
512
|
+
};
|
|
513
|
+
}
|
|
514
|
+
else {
|
|
515
|
+
pNode = {
|
|
516
|
+
type: 'paragraph',
|
|
517
|
+
text: pContent.text,
|
|
518
|
+
children: pContent.children,
|
|
519
|
+
metadata: {
|
|
520
|
+
...(pContent.alignment ? { alignment: pContent.alignment } : {}),
|
|
521
|
+
...(pContent.style ? { style: pContent.style } : {})
|
|
522
|
+
}
|
|
523
|
+
};
|
|
524
|
+
}
|
|
500
525
|
// Clean up metadata if empty
|
|
501
|
-
if (Object.keys(pNode.metadata || {}).length === 0)
|
|
526
|
+
if (pNode.type === 'paragraph' && Object.keys(pNode.metadata || {}).length === 0) {
|
|
502
527
|
delete pNode.metadata;
|
|
503
|
-
if (element.tagName === "text:h") {
|
|
504
|
-
if (!pNode.metadata)
|
|
505
|
-
pNode.metadata = {};
|
|
506
|
-
pNode.metadata.level = parseInt(element.getAttribute("text:outline-level") || "1");
|
|
507
528
|
}
|
|
508
529
|
if (config.includeRawContent) {
|
|
509
530
|
pNode.rawContent = (0, xmlUtils_js_1.getRawContent)(element, sourceXml, config);
|
|
@@ -535,11 +556,18 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
535
556
|
}
|
|
536
557
|
// Add cell(s) for repeated columns
|
|
537
558
|
for (let k = 0; k < colsRepeated; k++) {
|
|
559
|
+
// Apply cell background color if defined in styleMap
|
|
560
|
+
const cellStyleName = cell.getAttribute("table:style-name");
|
|
561
|
+
const cellBgColor = cellStyleName && styleMap[cellStyleName]?.backgroundColor;
|
|
538
562
|
const cellNode = {
|
|
539
563
|
type: 'cell',
|
|
540
564
|
text: cellText,
|
|
541
565
|
children: cellChildren.length > 0 ? (k === 0 ? cellChildren : JSON.parse(JSON.stringify(cellChildren))) : [],
|
|
542
|
-
metadata: {
|
|
566
|
+
metadata: {
|
|
567
|
+
row: rowIndex,
|
|
568
|
+
col: colIndex,
|
|
569
|
+
...(cellBgColor ? { backgroundColor: cellBgColor } : {})
|
|
570
|
+
}
|
|
543
571
|
};
|
|
544
572
|
const cellMetadata = cellNode.metadata;
|
|
545
573
|
if (colSpan > 1)
|
|
@@ -617,9 +645,15 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
617
645
|
if (Object.keys(styleInfo).length > 0) {
|
|
618
646
|
paragraphStyleMap[name] = styleInfo;
|
|
619
647
|
}
|
|
648
|
+
const cellProps = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "style:table-cell-properties");
|
|
649
|
+
const formatting = {};
|
|
650
|
+
if (cellProps) {
|
|
651
|
+
const bgColor = cellProps.getAttribute("fo:background-color");
|
|
652
|
+
if (bgColor && bgColor !== 'transparent')
|
|
653
|
+
formatting.backgroundColor = bgColor;
|
|
654
|
+
}
|
|
620
655
|
const textProps = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "style:text-properties");
|
|
621
656
|
if (textProps) {
|
|
622
|
-
const formatting = {};
|
|
623
657
|
if (textProps.getAttribute("fo:font-weight") === "bold" || textProps.getAttribute("style:font-weight-asian") === "bold")
|
|
624
658
|
formatting.bold = true;
|
|
625
659
|
if (textProps.getAttribute("fo:font-style") === "italic" || textProps.getAttribute("style:font-style-asian") === "italic")
|
|
@@ -650,9 +684,9 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
650
684
|
if (textPosition.startsWith("super"))
|
|
651
685
|
formatting.superscript = true;
|
|
652
686
|
}
|
|
653
|
-
if (Object.keys(formatting).length > 0)
|
|
654
|
-
styleMap[name] = formatting;
|
|
655
687
|
}
|
|
688
|
+
if (Object.keys(formatting).length > 0)
|
|
689
|
+
styleMap[name] = formatting;
|
|
656
690
|
}
|
|
657
691
|
}
|
|
658
692
|
// Start traversal
|
|
@@ -1267,12 +1301,9 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1267
1301
|
}
|
|
1268
1302
|
content.push(slideNode);
|
|
1269
1303
|
if (noteNode && noteNode.children && noteNode.children.length > 0) {
|
|
1270
|
-
if (
|
|
1271
|
-
|
|
1272
|
-
|
|
1273
|
-
else {
|
|
1274
|
-
content.push(noteNode);
|
|
1275
|
-
}
|
|
1304
|
+
if (!slideNode.notes)
|
|
1305
|
+
slideNode.notes = [];
|
|
1306
|
+
slideNode.notes.push(noteNode);
|
|
1276
1307
|
}
|
|
1277
1308
|
}
|
|
1278
1309
|
if (odpNotes.length > 0) {
|
|
@@ -1495,10 +1526,6 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1495
1526
|
};
|
|
1496
1527
|
}
|
|
1497
1528
|
}
|
|
1498
|
-
// Append notes to content if configured
|
|
1499
|
-
if (config.putNotesAtLast && notes.length > 0) {
|
|
1500
|
-
content.push(...notes);
|
|
1501
|
-
}
|
|
1502
1529
|
const toTextSync = () => content.map(c => {
|
|
1503
1530
|
const getText = (node) => {
|
|
1504
1531
|
let t = '';
|
|
@@ -1520,6 +1547,6 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1520
1547
|
return (0, astUtils_js_1.createAST)(fileType, {
|
|
1521
1548
|
...metadata,
|
|
1522
1549
|
styleMap: combinedStyleMap
|
|
1523
|
-
}, content, attachments, config, toTextSync);
|
|
1550
|
+
}, content, attachments, config, undefined, toTextSync);
|
|
1524
1551
|
};
|
|
1525
1552
|
exports.parseOpenOffice = parseOpenOffice;
|
|
@@ -266,6 +266,7 @@ function convertToRgbaBuffer(data, width, height, kind) {
|
|
|
266
266
|
* @returns Promise resolving to the parsed AST
|
|
267
267
|
*/
|
|
268
268
|
const parsePdf = async (buffer, config) => {
|
|
269
|
+
(0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
|
|
269
270
|
const pdfjs = await (0, moduleLoader_js_1.loadPdfJs)();
|
|
270
271
|
// Configure worker
|
|
271
272
|
const workerSrc = config.pdfWorkerSrc;
|
|
@@ -350,6 +351,17 @@ const parsePdf = async (buffer, config) => {
|
|
|
350
351
|
'IsCollectionPresent', 'IsSignaturesPresent', 'PDFFormatVersion'
|
|
351
352
|
]);
|
|
352
353
|
if (info) {
|
|
354
|
+
metadata.nativeProperties = {};
|
|
355
|
+
for (const [key, val] of Object.entries(info)) {
|
|
356
|
+
if (key === 'Custom' && typeof val === 'object' && !Array.isArray(val) && !(val instanceof Date) && val !== null) {
|
|
357
|
+
for (const [customKey, customVal] of Object.entries(val)) {
|
|
358
|
+
metadata.nativeProperties[customKey] = customVal;
|
|
359
|
+
}
|
|
360
|
+
}
|
|
361
|
+
else {
|
|
362
|
+
metadata.nativeProperties[key] = val;
|
|
363
|
+
}
|
|
364
|
+
}
|
|
353
365
|
const customProperties = {};
|
|
354
366
|
for (const key of Object.keys(info)) {
|
|
355
367
|
if (standardPdfInfoKeys.has(key))
|
|
@@ -377,6 +389,17 @@ const parsePdf = async (buffer, config) => {
|
|
|
377
389
|
metadata.customProperties = customProperties;
|
|
378
390
|
}
|
|
379
391
|
}
|
|
392
|
+
if (meta.metadata) {
|
|
393
|
+
if (!metadata.nativeProperties)
|
|
394
|
+
metadata.nativeProperties = {};
|
|
395
|
+
const xmp = meta.metadata;
|
|
396
|
+
if (typeof xmp.getAll === 'function') {
|
|
397
|
+
metadata.nativeProperties['XMP'] = xmp.getAll();
|
|
398
|
+
}
|
|
399
|
+
else {
|
|
400
|
+
metadata.nativeProperties['XMP'] = xmp;
|
|
401
|
+
}
|
|
402
|
+
}
|
|
380
403
|
// --- Embedded File Attachment Extraction ---
|
|
381
404
|
/**
|
|
382
405
|
* PDF can contain embedded files (not images in content, but attached files).
|
|
@@ -398,6 +421,7 @@ const parsePdf = async (buffer, config) => {
|
|
|
398
421
|
}
|
|
399
422
|
// --- First Pass: Collect all items for font statistics ---
|
|
400
423
|
for (let i = 1; i <= numPages; i++) {
|
|
424
|
+
(0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
|
|
401
425
|
let page;
|
|
402
426
|
let textContent;
|
|
403
427
|
const pageItems = [];
|
|
@@ -575,6 +599,7 @@ const parsePdf = async (buffer, config) => {
|
|
|
575
599
|
const fontStats = calculateFontStats(allPageItems);
|
|
576
600
|
// --- Second Pass: Process pages with font statistics ---
|
|
577
601
|
for (let i = 0; i < allPageItems.length; i++) {
|
|
602
|
+
(0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
|
|
578
603
|
const pageNum = i + 1;
|
|
579
604
|
let page;
|
|
580
605
|
try {
|
|
@@ -790,7 +815,7 @@ const parsePdf = async (buffer, config) => {
|
|
|
790
815
|
});
|
|
791
816
|
}
|
|
792
817
|
const toTextSync = () => content.map(c => c.text).join(config.newlineDelimiter);
|
|
793
|
-
return (0, astUtils_js_1.createAST)('pdf', metadata, content, attachments, config, toTextSync);
|
|
818
|
+
return (0, astUtils_js_1.createAST)('pdf', metadata, content, attachments, config, undefined, toTextSync);
|
|
794
819
|
};
|
|
795
820
|
exports.parsePdf = parsePdf;
|
|
796
821
|
/**
|