officeparser 7.1.0 → 7.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +152 -56
- package/dist/OfficeGenerator.d.ts +6 -2
- package/dist/OfficeGenerator.js +30 -9
- package/dist/OfficeParser.d.ts +1 -1
- package/dist/OfficeParser.js +1 -1
- package/dist/cli.d.ts +18 -12
- package/dist/cli.js +255 -81
- package/dist/defaults.js +12 -1
- package/dist/generators/BaseGenerator.d.ts +4 -3
- package/dist/generators/BaseGenerator.js +13 -1
- package/dist/generators/ChunkingGenerator.js +32 -5
- package/dist/generators/CsvGenerator.d.ts +1 -1
- package/dist/generators/HtmlGenerator.d.ts +2 -1
- package/dist/generators/HtmlGenerator.js +481 -42
- package/dist/generators/MarkdownGenerator.d.ts +1 -1
- package/dist/generators/MarkdownGenerator.js +35 -2
- package/dist/generators/PdfGenerator.d.ts +1 -1
- package/dist/generators/PdfGenerator.js +0 -6
- package/dist/generators/RtfGenerator.d.ts +2 -1
- package/dist/generators/RtfGenerator.js +49 -6
- package/dist/generators/TextGenerator.d.ts +1 -1
- package/dist/generators/TextGenerator.js +6 -0
- package/dist/officeparser.browser.d.ts +267 -54
- package/dist/officeparser.browser.iife.js +599 -187
- package/dist/officeparser.browser.mjs +599 -187
- package/dist/parsers/CsvParser.js +1 -1
- package/dist/parsers/ExcelParser.js +63 -19
- package/dist/parsers/HtmlParser.js +10 -1
- package/dist/parsers/MarkdownParser.js +13 -10
- package/dist/parsers/OpenOfficeParser.js +57 -34
- package/dist/parsers/PdfParser.js +28 -3
- package/dist/parsers/PowerPointParser.js +164 -40
- package/dist/parsers/RtfParser.js +28 -24
- package/dist/parsers/WordParser.js +154 -11
- package/dist/sbom.cdx.json +100 -100
- package/dist/types.d.ts +268 -53
- package/dist/types.js +4 -0
- package/dist/utils/astUtils.d.ts +2 -2
- package/dist/utils/astUtils.js +2 -1
- package/dist/utils/configUtils.d.ts +5 -0
- package/dist/utils/configUtils.js +55 -1
- package/dist/utils/errorUtils.js +3 -1
- package/dist/utils/moduleLoader.js +55 -11
- package/dist/utils/xmlUtils.d.ts +9 -0
- package/dist/utils/xmlUtils.js +53 -1
- package/package.json +6 -3
|
@@ -110,6 +110,6 @@ const parseCsv = async (buffer, config) => {
|
|
|
110
110
|
.join(config.newlineDelimiter)
|
|
111
111
|
.replace(/\n{3,}/g, '\n\n');
|
|
112
112
|
};
|
|
113
|
-
return (0, astUtils_js_1.createAST)('csv', { title: 'Sheet1' }, [sheetNode], [], config, toTextSync);
|
|
113
|
+
return (0, astUtils_js_1.createAST)('csv', { title: 'Sheet1' }, [sheetNode], [], config, undefined, toTextSync);
|
|
114
114
|
};
|
|
115
115
|
exports.parseCsv = parseCsv;
|
|
@@ -51,18 +51,23 @@ const parseExcel = async (buffer, config) => {
|
|
|
51
51
|
const mediaFileRegex = /xl\/media\/.*/;
|
|
52
52
|
const corePropsFileRegex = /docProps\/core\.xml/;
|
|
53
53
|
const customPropsFileRegex = /docProps\/custom\.xml/;
|
|
54
|
+
const appPropsFileRegex = /docProps\/app\.xml/;
|
|
54
55
|
const relsRegex = /xl\/worksheets\/_rels\/sheet\d+\.xml\.rels/g;
|
|
55
56
|
const drawingRelsRegex = /xl\/drawings\/_rels\/drawing\d+\.xml\.rels/g;
|
|
57
|
+
const commentsRegex = /xl\/comments\d+\.xml/g;
|
|
56
58
|
const files = await (0, zipUtils_js_1.extractFiles)(buffer, (x) => !!x.match(sheetsRegex) ||
|
|
57
59
|
!!x.match(drawingsRegex) ||
|
|
58
60
|
!!x.match(chartsRegex) ||
|
|
61
|
+
(!config.ignoreComments && !!x.match(commentsRegex)) ||
|
|
59
62
|
x === stringsFilePath ||
|
|
60
63
|
x === 'xl/styles.xml' ||
|
|
61
64
|
x === 'xl/workbook.xml' ||
|
|
62
65
|
x === 'xl/_rels/workbook.xml.rels' ||
|
|
63
66
|
!!x.match(corePropsFileRegex) ||
|
|
64
67
|
!!x.match(customPropsFileRegex) ||
|
|
65
|
-
|
|
68
|
+
!!x.match(appPropsFileRegex) ||
|
|
69
|
+
(!!config.extractAttachments && (!!x.match(mediaFileRegex) || !!x.match(drawingRelsRegex))) ||
|
|
70
|
+
((!!config.extractAttachments || !config.ignoreComments) && !!x.match(relsRegex)));
|
|
66
71
|
const sharedStringsFile = files.find(f => f.path === stringsFilePath);
|
|
67
72
|
// Updated to store structured content (rich text runs) or simple string
|
|
68
73
|
const sharedStrings = [];
|
|
@@ -415,6 +420,52 @@ const parseExcel = async (buffer, config) => {
|
|
|
415
420
|
if (config.includeRawContent) {
|
|
416
421
|
rawContents.push(file.content.toString());
|
|
417
422
|
}
|
|
423
|
+
const sheetFilename = file.path.split('/').pop() || '';
|
|
424
|
+
const relsFilename = `xl/worksheets/_rels/${sheetFilename}.rels`;
|
|
425
|
+
const relsFile = files.find(f => f.path === relsFilename);
|
|
426
|
+
const drawingMap = {}; // rId -> drawingPath
|
|
427
|
+
const sheetCommentsMap = {};
|
|
428
|
+
if (relsFile) {
|
|
429
|
+
const relsXml = (0, xmlUtils_js_1.parseXmlString)(relsFile.content.toString());
|
|
430
|
+
const relationships = (0, xmlUtils_js_1.getElementsByTagName)(relsXml, "Relationship");
|
|
431
|
+
for (const rel of relationships) {
|
|
432
|
+
const id = rel.getAttribute("Id");
|
|
433
|
+
const target = rel.getAttribute("Target");
|
|
434
|
+
const type = rel.getAttribute("Type");
|
|
435
|
+
if (id && target && type) {
|
|
436
|
+
if (config.extractAttachments && type.includes('drawing')) {
|
|
437
|
+
drawingMap[id] = 'xl/drawings/' + target.replace('../drawings/', '');
|
|
438
|
+
}
|
|
439
|
+
else if (!config.ignoreComments && type.includes('comments')) {
|
|
440
|
+
const commentsPath = 'xl/' + target.replace('../', '');
|
|
441
|
+
const cFile = files.find(f => f.path === commentsPath);
|
|
442
|
+
if (cFile) {
|
|
443
|
+
const cXml = (0, xmlUtils_js_1.parseXmlString)(cFile.content.toString());
|
|
444
|
+
const commentNodes = (0, xmlUtils_js_1.getElementsByTagName)(cXml, "comment");
|
|
445
|
+
const authorsList = (0, xmlUtils_js_1.getElementsByTagName)(cXml, "author");
|
|
446
|
+
const authors = authorsList.map(a => a.textContent || '');
|
|
447
|
+
for (const cNode of commentNodes) {
|
|
448
|
+
const ref = cNode.getAttribute("ref");
|
|
449
|
+
const authorId = cNode.getAttribute("authorId");
|
|
450
|
+
const author = authorId !== null ? authors[parseInt(authorId)] : undefined;
|
|
451
|
+
const tNodes = (0, xmlUtils_js_1.getElementsByTagName)(cNode, "t");
|
|
452
|
+
const text = tNodes.map(t => t.textContent || '').join('');
|
|
453
|
+
if (ref && text) {
|
|
454
|
+
if (!sheetCommentsMap[ref])
|
|
455
|
+
sheetCommentsMap[ref] = [];
|
|
456
|
+
sheetCommentsMap[ref].push({
|
|
457
|
+
type: 'comment',
|
|
458
|
+
text: text,
|
|
459
|
+
children: [{ type: 'text', text: text, formatting: {} }],
|
|
460
|
+
metadata: author ? { author } : undefined
|
|
461
|
+
});
|
|
462
|
+
}
|
|
463
|
+
}
|
|
464
|
+
}
|
|
465
|
+
}
|
|
466
|
+
}
|
|
467
|
+
}
|
|
468
|
+
}
|
|
418
469
|
const rows = [];
|
|
419
470
|
const sheetXml = file.content.toString();
|
|
420
471
|
// regex to match <row> elements, capturing:
|
|
@@ -485,7 +536,9 @@ const parseExcel = async (buffer, config) => {
|
|
|
485
536
|
// Parse cell coordinate
|
|
486
537
|
const coordMatch = cAttrs.match(/r="([A-Z]+)(\d+)"/);
|
|
487
538
|
let colIndex;
|
|
539
|
+
let ref;
|
|
488
540
|
if (coordMatch) {
|
|
541
|
+
ref = coordMatch[1] + coordMatch[2];
|
|
489
542
|
colIndex = colToNumber(coordMatch[1]);
|
|
490
543
|
// If row index is missing in cell coord (unlikely but possible), use rowIndex
|
|
491
544
|
}
|
|
@@ -525,10 +578,12 @@ const parseExcel = async (buffer, config) => {
|
|
|
525
578
|
formatting: cellFormatting
|
|
526
579
|
});
|
|
527
580
|
}
|
|
581
|
+
const commentsNodeList = (ref && sheetCommentsMap[ref]) ? sheetCommentsMap[ref] : undefined;
|
|
528
582
|
const cellNode = {
|
|
529
583
|
type: 'cell',
|
|
530
584
|
text: text,
|
|
531
585
|
children: cellNodes,
|
|
586
|
+
comments: commentsNodeList,
|
|
532
587
|
metadata: { row: rowIndex, col: colIndex }
|
|
533
588
|
};
|
|
534
589
|
if (config.includeRawContent) {
|
|
@@ -551,23 +606,6 @@ const parseExcel = async (buffer, config) => {
|
|
|
551
606
|
}
|
|
552
607
|
// Handle Drawings in Sheet (images and charts)
|
|
553
608
|
if (config.extractAttachments) {
|
|
554
|
-
// Parse Sheet Rels to map drawing rIds
|
|
555
|
-
const sheetFilename = file.path.split('/').pop() || '';
|
|
556
|
-
const relsFilename = `xl/worksheets/_rels/${sheetFilename}.rels`;
|
|
557
|
-
const relsFile = files.find(f => f.path === relsFilename);
|
|
558
|
-
const drawingMap = {}; // rId -> drawingPath
|
|
559
|
-
if (relsFile) {
|
|
560
|
-
const relsXml = (0, xmlUtils_js_1.parseXmlString)(relsFile.content.toString());
|
|
561
|
-
const relationships = (0, xmlUtils_js_1.getElementsByTagName)(relsXml, "Relationship");
|
|
562
|
-
for (const rel of relationships) {
|
|
563
|
-
const id = rel.getAttribute("Id");
|
|
564
|
-
const target = rel.getAttribute("Target");
|
|
565
|
-
const type = rel.getAttribute("Type");
|
|
566
|
-
if (id && target && type && type.includes('drawing')) {
|
|
567
|
-
drawingMap[id] = 'xl/drawings/' + target.replace('../drawings/', '');
|
|
568
|
-
}
|
|
569
|
-
}
|
|
570
|
-
}
|
|
571
609
|
const drawingMatches = file.content.toString().match(/<drawing r:id="(.*?)"/g);
|
|
572
610
|
if (drawingMatches) {
|
|
573
611
|
for (const match of drawingMatches) {
|
|
@@ -637,6 +675,12 @@ const parseExcel = async (buffer, config) => {
|
|
|
637
675
|
if (Object.keys(customProperties).length > 0)
|
|
638
676
|
metadata.customProperties = customProperties;
|
|
639
677
|
}
|
|
678
|
+
const appPropsFile = files.find(f => f.path.match(appPropsFileRegex));
|
|
679
|
+
if (appPropsFile) {
|
|
680
|
+
const appProperties = (0, xmlUtils_js_1.parseOOXMLAppProperties)(appPropsFile.content.toString());
|
|
681
|
+
if (Object.keys(appProperties).length > 0)
|
|
682
|
+
metadata.nativeProperties = appProperties;
|
|
683
|
+
}
|
|
640
684
|
// Link OCR text and chart data to content nodes (like PPTX parser)
|
|
641
685
|
const assignAttachmentData = (nodes) => {
|
|
642
686
|
for (const node of nodes) {
|
|
@@ -681,6 +725,6 @@ const parseExcel = async (buffer, config) => {
|
|
|
681
725
|
};
|
|
682
726
|
return getText(c);
|
|
683
727
|
}).filter(t => t != '').join(config.newlineDelimiter);
|
|
684
|
-
return (0, astUtils_js_1.createAST)('xlsx', metadata, content, attachments, config, toTextSync);
|
|
728
|
+
return (0, astUtils_js_1.createAST)('xlsx', metadata, content, attachments, config, undefined, toTextSync);
|
|
685
729
|
};
|
|
686
730
|
exports.parseExcel = parseExcel;
|
|
@@ -127,6 +127,15 @@ const parseHtml = async (buffer, config) => {
|
|
|
127
127
|
if (titleNode && titleNode.children.length > 0 && titleNode.children[0].text) {
|
|
128
128
|
metadata.title = titleNode.children[0].text;
|
|
129
129
|
}
|
|
130
|
+
metadata.nativeProperties = {};
|
|
131
|
+
for (const child of head.children) {
|
|
132
|
+
if (child.tagName === 'meta') {
|
|
133
|
+
const name = child.attributes?.name || child.attributes?.property || child.attributes?.['http-equiv'];
|
|
134
|
+
if (name) {
|
|
135
|
+
metadata.nativeProperties[name] = child.attributes?.content || '';
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
}
|
|
130
139
|
const extractMeta = (name) => {
|
|
131
140
|
for (const child of head.children) {
|
|
132
141
|
if (child.tagName === 'meta' && (child.attributes?.name === name || child.attributes?.property === name)) {
|
|
@@ -539,6 +548,6 @@ const parseHtml = async (buffer, config) => {
|
|
|
539
548
|
return getText(n);
|
|
540
549
|
}).join(config.newlineDelimiter)
|
|
541
550
|
.replace(/\n{3,}/g, '\n\n'); // Normalize excessive whitespace
|
|
542
|
-
return (0, astUtils_js_1.createAST)('html', metadata, content, attachments, config, toTextSync);
|
|
551
|
+
return (0, astUtils_js_1.createAST)('html', metadata, content, attachments, config, undefined, toTextSync);
|
|
543
552
|
};
|
|
544
553
|
exports.parseHtml = parseHtml;
|
|
@@ -21,11 +21,20 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
21
21
|
textStr = textStr.substring(endIdx + 5);
|
|
22
22
|
const lines = frontMatter.split('\n');
|
|
23
23
|
const customProps = {};
|
|
24
|
+
const nativeProps = {};
|
|
24
25
|
for (const line of lines) {
|
|
25
26
|
const match = line.match(/^([^:]+):\s*(.*)$/);
|
|
26
27
|
if (match) {
|
|
27
28
|
const key = match[1].trim();
|
|
28
29
|
let val = match[2].trim().replace(/^"(.*)"$/, '$1');
|
|
30
|
+
let parsedVal = val;
|
|
31
|
+
if (val === 'true')
|
|
32
|
+
parsedVal = true;
|
|
33
|
+
else if (val === 'false')
|
|
34
|
+
parsedVal = false;
|
|
35
|
+
else if (!isNaN(Number(val)) && val !== '')
|
|
36
|
+
parsedVal = Number(val);
|
|
37
|
+
nativeProps[key] = parsedVal;
|
|
29
38
|
if (key === 'title')
|
|
30
39
|
metadata.title = val;
|
|
31
40
|
else if (key === 'author')
|
|
@@ -37,20 +46,14 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
37
46
|
else if (key === 'description')
|
|
38
47
|
metadata.description = val;
|
|
39
48
|
else {
|
|
40
|
-
|
|
41
|
-
if (val === 'true')
|
|
42
|
-
customProps[key] = true;
|
|
43
|
-
else if (val === 'false')
|
|
44
|
-
customProps[key] = false;
|
|
45
|
-
else if (!isNaN(Number(val)) && val !== '')
|
|
46
|
-
customProps[key] = Number(val);
|
|
47
|
-
else
|
|
48
|
-
customProps[key] = val;
|
|
49
|
+
customProps[key] = parsedVal;
|
|
49
50
|
}
|
|
50
51
|
}
|
|
51
52
|
}
|
|
52
53
|
if (Object.keys(customProps).length > 0)
|
|
53
54
|
metadata.customProperties = customProps;
|
|
55
|
+
if (Object.keys(nativeProps).length > 0)
|
|
56
|
+
metadata.nativeProperties = nativeProps;
|
|
54
57
|
}
|
|
55
58
|
}
|
|
56
59
|
// Extract code blocks first to protect their contents
|
|
@@ -360,6 +363,6 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
360
363
|
return getText(n);
|
|
361
364
|
}).join(config.newlineDelimiter)
|
|
362
365
|
.replace(/\n{3,}/g, '\n\n'); // Normalize excessive whitespace
|
|
363
|
-
return (0, astUtils_js_1.createAST)('md', metadata, content, attachments, config, toTextSync);
|
|
366
|
+
return (0, astUtils_js_1.createAST)('md', metadata, content, attachments, config, undefined, toTextSync);
|
|
364
367
|
};
|
|
365
368
|
exports.parseMarkdown = parseMarkdown;
|
|
@@ -157,9 +157,9 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
157
157
|
if (textPosition.startsWith("super"))
|
|
158
158
|
formatting.superscript = true;
|
|
159
159
|
}
|
|
160
|
-
if (Object.keys(formatting).length > 0)
|
|
161
|
-
styleMap[name] = formatting;
|
|
162
160
|
}
|
|
161
|
+
if (Object.keys(formatting).length > 0)
|
|
162
|
+
styleMap[name] = formatting;
|
|
163
163
|
}
|
|
164
164
|
};
|
|
165
165
|
if (stylesDom) {
|
|
@@ -310,11 +310,17 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
310
310
|
noteId: noteId
|
|
311
311
|
}
|
|
312
312
|
};
|
|
313
|
-
if (
|
|
314
|
-
|
|
313
|
+
if (children.length > 0 && children[children.length - 1].type === 'text') {
|
|
314
|
+
const precedingNode = children[children.length - 1];
|
|
315
|
+
if (!precedingNode.notes) {
|
|
316
|
+
precedingNode.notes = [];
|
|
317
|
+
}
|
|
318
|
+
precedingNode.notes.push(noteNode);
|
|
315
319
|
}
|
|
316
320
|
else {
|
|
317
|
-
|
|
321
|
+
const emptyTextNode = { type: 'text', text: '' };
|
|
322
|
+
emptyTextNode.notes = [noteNode];
|
|
323
|
+
children.push(emptyTextNode);
|
|
318
324
|
}
|
|
319
325
|
}
|
|
320
326
|
}
|
|
@@ -492,22 +498,33 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
492
498
|
const element = child;
|
|
493
499
|
if (element.tagName === "text:p" || element.tagName === "text:h") {
|
|
494
500
|
const pContent = parseParagraphContent(element, paraStyleMap, styleMap, config, sourceXml);
|
|
495
|
-
|
|
496
|
-
|
|
497
|
-
|
|
498
|
-
|
|
499
|
-
|
|
500
|
-
|
|
501
|
-
|
|
502
|
-
|
|
503
|
-
|
|
501
|
+
let pNode;
|
|
502
|
+
if (element.tagName === "text:h") {
|
|
503
|
+
pNode = {
|
|
504
|
+
type: 'heading',
|
|
505
|
+
text: pContent.text,
|
|
506
|
+
children: pContent.children,
|
|
507
|
+
metadata: {
|
|
508
|
+
level: parseInt(element.getAttribute("text:outline-level") || "1"),
|
|
509
|
+
...(pContent.alignment ? { alignment: pContent.alignment } : {}),
|
|
510
|
+
...(pContent.style ? { style: pContent.style } : {})
|
|
511
|
+
}
|
|
512
|
+
};
|
|
513
|
+
}
|
|
514
|
+
else {
|
|
515
|
+
pNode = {
|
|
516
|
+
type: 'paragraph',
|
|
517
|
+
text: pContent.text,
|
|
518
|
+
children: pContent.children,
|
|
519
|
+
metadata: {
|
|
520
|
+
...(pContent.alignment ? { alignment: pContent.alignment } : {}),
|
|
521
|
+
...(pContent.style ? { style: pContent.style } : {})
|
|
522
|
+
}
|
|
523
|
+
};
|
|
524
|
+
}
|
|
504
525
|
// Clean up metadata if empty
|
|
505
|
-
if (Object.keys(pNode.metadata || {}).length === 0)
|
|
526
|
+
if (pNode.type === 'paragraph' && Object.keys(pNode.metadata || {}).length === 0) {
|
|
506
527
|
delete pNode.metadata;
|
|
507
|
-
if (element.tagName === "text:h") {
|
|
508
|
-
if (!pNode.metadata)
|
|
509
|
-
pNode.metadata = {};
|
|
510
|
-
pNode.metadata.level = parseInt(element.getAttribute("text:outline-level") || "1");
|
|
511
528
|
}
|
|
512
529
|
if (config.includeRawContent) {
|
|
513
530
|
pNode.rawContent = (0, xmlUtils_js_1.getRawContent)(element, sourceXml, config);
|
|
@@ -539,11 +556,18 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
539
556
|
}
|
|
540
557
|
// Add cell(s) for repeated columns
|
|
541
558
|
for (let k = 0; k < colsRepeated; k++) {
|
|
559
|
+
// Apply cell background color if defined in styleMap
|
|
560
|
+
const cellStyleName = cell.getAttribute("table:style-name");
|
|
561
|
+
const cellBgColor = cellStyleName && styleMap[cellStyleName]?.backgroundColor;
|
|
542
562
|
const cellNode = {
|
|
543
563
|
type: 'cell',
|
|
544
564
|
text: cellText,
|
|
545
565
|
children: cellChildren.length > 0 ? (k === 0 ? cellChildren : JSON.parse(JSON.stringify(cellChildren))) : [],
|
|
546
|
-
metadata: {
|
|
566
|
+
metadata: {
|
|
567
|
+
row: rowIndex,
|
|
568
|
+
col: colIndex,
|
|
569
|
+
...(cellBgColor ? { backgroundColor: cellBgColor } : {})
|
|
570
|
+
}
|
|
547
571
|
};
|
|
548
572
|
const cellMetadata = cellNode.metadata;
|
|
549
573
|
if (colSpan > 1)
|
|
@@ -621,9 +645,15 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
621
645
|
if (Object.keys(styleInfo).length > 0) {
|
|
622
646
|
paragraphStyleMap[name] = styleInfo;
|
|
623
647
|
}
|
|
648
|
+
const cellProps = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "style:table-cell-properties");
|
|
649
|
+
const formatting = {};
|
|
650
|
+
if (cellProps) {
|
|
651
|
+
const bgColor = cellProps.getAttribute("fo:background-color");
|
|
652
|
+
if (bgColor && bgColor !== 'transparent')
|
|
653
|
+
formatting.backgroundColor = bgColor;
|
|
654
|
+
}
|
|
624
655
|
const textProps = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "style:text-properties");
|
|
625
656
|
if (textProps) {
|
|
626
|
-
const formatting = {};
|
|
627
657
|
if (textProps.getAttribute("fo:font-weight") === "bold" || textProps.getAttribute("style:font-weight-asian") === "bold")
|
|
628
658
|
formatting.bold = true;
|
|
629
659
|
if (textProps.getAttribute("fo:font-style") === "italic" || textProps.getAttribute("style:font-style-asian") === "italic")
|
|
@@ -654,9 +684,9 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
654
684
|
if (textPosition.startsWith("super"))
|
|
655
685
|
formatting.superscript = true;
|
|
656
686
|
}
|
|
657
|
-
if (Object.keys(formatting).length > 0)
|
|
658
|
-
styleMap[name] = formatting;
|
|
659
687
|
}
|
|
688
|
+
if (Object.keys(formatting).length > 0)
|
|
689
|
+
styleMap[name] = formatting;
|
|
660
690
|
}
|
|
661
691
|
}
|
|
662
692
|
// Start traversal
|
|
@@ -1271,12 +1301,9 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1271
1301
|
}
|
|
1272
1302
|
content.push(slideNode);
|
|
1273
1303
|
if (noteNode && noteNode.children && noteNode.children.length > 0) {
|
|
1274
|
-
if (
|
|
1275
|
-
|
|
1276
|
-
|
|
1277
|
-
else {
|
|
1278
|
-
content.push(noteNode);
|
|
1279
|
-
}
|
|
1304
|
+
if (!slideNode.notes)
|
|
1305
|
+
slideNode.notes = [];
|
|
1306
|
+
slideNode.notes.push(noteNode);
|
|
1280
1307
|
}
|
|
1281
1308
|
}
|
|
1282
1309
|
if (odpNotes.length > 0) {
|
|
@@ -1499,10 +1526,6 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1499
1526
|
};
|
|
1500
1527
|
}
|
|
1501
1528
|
}
|
|
1502
|
-
// Append notes to content if configured
|
|
1503
|
-
if (config.putNotesAtLast && notes.length > 0) {
|
|
1504
|
-
content.push(...notes);
|
|
1505
|
-
}
|
|
1506
1529
|
const toTextSync = () => content.map(c => {
|
|
1507
1530
|
const getText = (node) => {
|
|
1508
1531
|
let t = '';
|
|
@@ -1524,6 +1547,6 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1524
1547
|
return (0, astUtils_js_1.createAST)(fileType, {
|
|
1525
1548
|
...metadata,
|
|
1526
1549
|
styleMap: combinedStyleMap
|
|
1527
|
-
}, content, attachments, config, toTextSync);
|
|
1550
|
+
}, content, attachments, config, undefined, toTextSync);
|
|
1528
1551
|
};
|
|
1529
1552
|
exports.parseOpenOffice = parseOpenOffice;
|
|
@@ -274,15 +274,18 @@ const parsePdf = async (buffer, config) => {
|
|
|
274
274
|
pdfjs.GlobalWorkerOptions.workerSrc = workerSrc;
|
|
275
275
|
}
|
|
276
276
|
else {
|
|
277
|
-
// Node.js: Try to auto-resolve local worker path to avoid remote
|
|
277
|
+
// Node.js: Try to auto-resolve local worker path to avoid remote download errors
|
|
278
278
|
(0, envUtils_js_1.assertNode)('pdf-worker-auto-resolution');
|
|
279
279
|
let resolved = false;
|
|
280
280
|
// If the user provided a custom path (not the default CDN one), use it.
|
|
281
|
-
// Otherwise, try to find it locally.
|
|
281
|
+
// Otherwise, check if the worker is already loaded globally, or try to find it locally.
|
|
282
282
|
if (workerSrc !== defaults_js_1.DEFAULT_OFFICE_PARSER_CONFIG.pdfWorkerSrc && workerSrc !== '') {
|
|
283
283
|
pdfjs.GlobalWorkerOptions.workerSrc = workerSrc;
|
|
284
284
|
resolved = true;
|
|
285
285
|
}
|
|
286
|
+
else if (globalThis.pdfjsWorker) {
|
|
287
|
+
resolved = true;
|
|
288
|
+
}
|
|
286
289
|
else {
|
|
287
290
|
try {
|
|
288
291
|
// We use require.resolve to find the exact path of the installed package.
|
|
@@ -351,6 +354,17 @@ const parsePdf = async (buffer, config) => {
|
|
|
351
354
|
'IsCollectionPresent', 'IsSignaturesPresent', 'PDFFormatVersion'
|
|
352
355
|
]);
|
|
353
356
|
if (info) {
|
|
357
|
+
metadata.nativeProperties = {};
|
|
358
|
+
for (const [key, val] of Object.entries(info)) {
|
|
359
|
+
if (key === 'Custom' && typeof val === 'object' && !Array.isArray(val) && !(val instanceof Date) && val !== null) {
|
|
360
|
+
for (const [customKey, customVal] of Object.entries(val)) {
|
|
361
|
+
metadata.nativeProperties[customKey] = customVal;
|
|
362
|
+
}
|
|
363
|
+
}
|
|
364
|
+
else {
|
|
365
|
+
metadata.nativeProperties[key] = val;
|
|
366
|
+
}
|
|
367
|
+
}
|
|
354
368
|
const customProperties = {};
|
|
355
369
|
for (const key of Object.keys(info)) {
|
|
356
370
|
if (standardPdfInfoKeys.has(key))
|
|
@@ -378,6 +392,17 @@ const parsePdf = async (buffer, config) => {
|
|
|
378
392
|
metadata.customProperties = customProperties;
|
|
379
393
|
}
|
|
380
394
|
}
|
|
395
|
+
if (meta.metadata) {
|
|
396
|
+
if (!metadata.nativeProperties)
|
|
397
|
+
metadata.nativeProperties = {};
|
|
398
|
+
const xmp = meta.metadata;
|
|
399
|
+
if (typeof xmp.getAll === 'function') {
|
|
400
|
+
metadata.nativeProperties['XMP'] = xmp.getAll();
|
|
401
|
+
}
|
|
402
|
+
else {
|
|
403
|
+
metadata.nativeProperties['XMP'] = xmp;
|
|
404
|
+
}
|
|
405
|
+
}
|
|
381
406
|
// --- Embedded File Attachment Extraction ---
|
|
382
407
|
/**
|
|
383
408
|
* PDF can contain embedded files (not images in content, but attached files).
|
|
@@ -793,7 +818,7 @@ const parsePdf = async (buffer, config) => {
|
|
|
793
818
|
});
|
|
794
819
|
}
|
|
795
820
|
const toTextSync = () => content.map(c => c.text).join(config.newlineDelimiter);
|
|
796
|
-
return (0, astUtils_js_1.createAST)('pdf', metadata, content, attachments, config, toTextSync);
|
|
821
|
+
return (0, astUtils_js_1.createAST)('pdf', metadata, content, attachments, config, undefined, toTextSync);
|
|
797
822
|
};
|
|
798
823
|
exports.parsePdf = parsePdf;
|
|
799
824
|
/**
|