officeparser 6.1.1 → 7.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +301 -26
- package/dist/OfficeConverter.d.ts +46 -0
- package/dist/OfficeConverter.js +72 -0
- package/dist/OfficeGenerator.d.ts +19 -0
- package/dist/OfficeGenerator.js +48 -0
- package/dist/OfficeParser.d.ts +6 -0
- package/dist/OfficeParser.js +74 -31
- package/dist/cli.d.ts +3 -1
- package/dist/cli.js +106 -22
- package/dist/defaults.d.ts +41 -0
- package/dist/defaults.js +172 -0
- package/dist/generators/BaseGenerator.d.ts +58 -0
- package/dist/generators/BaseGenerator.js +107 -0
- package/dist/generators/ChunkingGenerator.d.ts +81 -0
- package/dist/generators/ChunkingGenerator.js +683 -0
- package/dist/generators/CsvGenerator.d.ts +30 -0
- package/dist/generators/CsvGenerator.js +233 -0
- package/dist/generators/HtmlGenerator.d.ts +37 -0
- package/dist/generators/HtmlGenerator.js +1013 -0
- package/dist/generators/MarkdownGenerator.d.ts +59 -0
- package/dist/generators/MarkdownGenerator.js +481 -0
- package/dist/generators/PdfGenerator.d.ts +22 -0
- package/dist/generators/PdfGenerator.js +118 -0
- package/dist/generators/RtfGenerator.d.ts +15 -0
- package/dist/generators/RtfGenerator.js +208 -0
- package/dist/generators/TextGenerator.d.ts +13 -0
- package/dist/generators/TextGenerator.js +108 -0
- package/dist/index.d.ts +11 -3
- package/dist/index.js +17 -2
- package/dist/index.mjs +2 -2
- package/dist/officeparser.browser.d.ts +828 -5
- package/dist/officeparser.browser.iife.js +703 -52
- package/dist/officeparser.browser.mjs +703 -52
- package/dist/parsers/CsvParser.d.ts +9 -0
- package/dist/parsers/CsvParser.js +110 -0
- package/dist/parsers/ExcelParser.d.ts +2 -2
- package/dist/parsers/ExcelParser.js +145 -114
- package/dist/parsers/HtmlParser.d.ts +2 -0
- package/dist/parsers/HtmlParser.js +539 -0
- package/dist/parsers/MarkdownParser.d.ts +2 -0
- package/dist/parsers/MarkdownParser.js +360 -0
- package/dist/parsers/OpenOfficeParser.d.ts +2 -2
- package/dist/parsers/OpenOfficeParser.js +140 -79
- package/dist/parsers/PdfParser.d.ts +2 -2
- package/dist/parsers/PdfParser.js +52 -49
- package/dist/parsers/PowerPointParser.d.ts +2 -2
- package/dist/parsers/PowerPointParser.js +20 -23
- package/dist/parsers/RtfParser.d.ts +2 -2
- package/dist/parsers/RtfParser.js +1291 -1240
- package/dist/parsers/WordParser.d.ts +2 -2
- package/dist/parsers/WordParser.js +232 -97
- package/dist/sbom.cdx.json +99 -99
- package/dist/types.d.ts +783 -5
- package/dist/types.js +73 -0
- package/dist/utils/astUtils.d.ts +16 -0
- package/dist/utils/astUtils.js +32 -0
- package/dist/utils/configUtils.d.ts +26 -0
- package/dist/utils/configUtils.js +140 -0
- package/dist/utils/envUtils.d.ts +8 -3
- package/dist/utils/envUtils.js +117 -34
- package/dist/utils/errorUtils.d.ts +17 -29
- package/dist/utils/errorUtils.js +110 -52
- package/dist/utils/moduleLoader.js +19 -11
- package/dist/utils/ocrUtils.js +2 -1
- package/dist/utils/sheetUtils.d.ts +7 -0
- package/dist/utils/sheetUtils.js +35 -0
- package/dist/utils/styleMapper.d.ts +36 -0
- package/dist/utils/styleMapper.js +224 -0
- package/dist/utils/xmlUtils.d.ts +0 -8
- package/dist/utils/xmlUtils.js +2 -1
- package/package.json +26 -7
|
@@ -23,6 +23,8 @@
|
|
|
23
23
|
*/
|
|
24
24
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
25
25
|
exports.parseOpenOffice = void 0;
|
|
26
|
+
const types_js_1 = require("../types.js");
|
|
27
|
+
const astUtils_js_1 = require("../utils/astUtils.js");
|
|
26
28
|
const chartUtils_js_1 = require("../utils/chartUtils.js");
|
|
27
29
|
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
28
30
|
const imageUtils_js_1 = require("../utils/imageUtils.js");
|
|
@@ -63,6 +65,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
63
65
|
}
|
|
64
66
|
const mainContentFile = files.find(f => f.path === 'content.xml') || files.find(f => f.path.match(contentFileRegex));
|
|
65
67
|
const stylesFile = files.find(f => f.path === 'styles.xml');
|
|
68
|
+
const stylesDom = stylesFile ? (0, xmlUtils_js_1.parseXmlString)(stylesFile.content.toString()) : undefined;
|
|
66
69
|
const content = [];
|
|
67
70
|
const notes = [];
|
|
68
71
|
// Style Map: styleName -> TextFormatting
|
|
@@ -76,8 +79,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
76
79
|
let listIdCounter = 0;
|
|
77
80
|
let lastWasList = false;
|
|
78
81
|
// Helper to parse styles
|
|
79
|
-
const parseStyles = (
|
|
80
|
-
const xml = (0, xmlUtils_js_1.parseXmlString)(xmlString);
|
|
82
|
+
const parseStyles = (xml) => {
|
|
81
83
|
const styles = (0, xmlUtils_js_1.getElementsByTagName)(xml, "style:style");
|
|
82
84
|
for (const style of styles) {
|
|
83
85
|
const name = style.getAttribute("style:name");
|
|
@@ -156,8 +158,8 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
156
158
|
}
|
|
157
159
|
}
|
|
158
160
|
};
|
|
159
|
-
if (
|
|
160
|
-
parseStyles(
|
|
161
|
+
if (stylesDom) {
|
|
162
|
+
parseStyles(stylesDom);
|
|
161
163
|
}
|
|
162
164
|
/**
|
|
163
165
|
* Helper to parse a paragraph node (text:p or text:h) and extract its content.
|
|
@@ -177,9 +179,10 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
177
179
|
*/
|
|
178
180
|
const parseInlineContent = (node, styleMap, config, notes, paragraphStyleMap, parentFormatting = {}, linkMetadata, sourceXml = '') => {
|
|
179
181
|
const children = [];
|
|
182
|
+
const anchorIds = [];
|
|
180
183
|
let fullText = '';
|
|
181
184
|
if (!node.childNodes)
|
|
182
|
-
return { text: '', children: [] };
|
|
185
|
+
return { text: '', children: [], anchorIds: [] };
|
|
183
186
|
for (let i = 0; i < node.childNodes.length; i++) {
|
|
184
187
|
const child = node.childNodes[i];
|
|
185
188
|
if (child.nodeType === 3) { // Text node
|
|
@@ -197,7 +200,12 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
197
200
|
else if ((0, xmlUtils_js_1.isElement)(child)) {
|
|
198
201
|
const element = child;
|
|
199
202
|
const tagName = element.tagName;
|
|
200
|
-
if (tagName === 'text:
|
|
203
|
+
if (tagName === 'text:bookmark' || tagName === 'text:bookmark-start') {
|
|
204
|
+
const name = element.getAttribute('text:name');
|
|
205
|
+
if (name)
|
|
206
|
+
anchorIds.push(name);
|
|
207
|
+
}
|
|
208
|
+
else if (tagName === 'text:s') {
|
|
201
209
|
// Space
|
|
202
210
|
const count = parseInt(element.getAttribute('text:c') || '1');
|
|
203
211
|
const spaces = ' '.repeat(count);
|
|
@@ -236,15 +244,34 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
236
244
|
const spanContent = parseInlineContent(element, styleMap, config, notes, paragraphStyleMap, formatting, linkMetadata, sourceXml);
|
|
237
245
|
fullText += spanContent.text;
|
|
238
246
|
children.push(...spanContent.children);
|
|
247
|
+
anchorIds.push(...spanContent.anchorIds);
|
|
239
248
|
}
|
|
240
249
|
else if (tagName === 'text:a') {
|
|
241
250
|
// Hyperlink
|
|
242
|
-
|
|
243
|
-
const
|
|
244
|
-
const
|
|
251
|
+
let href = element.getAttribute('xlink:href') || '';
|
|
252
|
+
const isInternal = href.startsWith('#');
|
|
253
|
+
const linkType = isInternal ? 'internal' : 'external';
|
|
254
|
+
if (isInternal) {
|
|
255
|
+
// ODT internal links can be encoded and might have suffixes like |outline
|
|
256
|
+
try {
|
|
257
|
+
href = decodeURIComponent(href).split('|')[0];
|
|
258
|
+
}
|
|
259
|
+
catch (e) {
|
|
260
|
+
href = href.split('|')[0];
|
|
261
|
+
}
|
|
262
|
+
// Normalize internal link: if it contains #, keep only from # onwards
|
|
263
|
+
if (href.includes('#')) {
|
|
264
|
+
href = '#' + href.split('#').pop();
|
|
265
|
+
}
|
|
266
|
+
}
|
|
267
|
+
let newLinkMetadata;
|
|
268
|
+
if (!isInternal || !config.ignoreInternalLinks) {
|
|
269
|
+
newLinkMetadata = { link: href, linkType: linkType };
|
|
270
|
+
}
|
|
245
271
|
const linkContent = parseInlineContent(element, styleMap, config, notes, paragraphStyleMap, parentFormatting, newLinkMetadata, sourceXml);
|
|
246
272
|
fullText += linkContent.text;
|
|
247
273
|
children.push(...linkContent.children);
|
|
274
|
+
anchorIds.push(...linkContent.anchorIds);
|
|
248
275
|
}
|
|
249
276
|
else if (tagName === 'text:note' && !config.ignoreNotes) {
|
|
250
277
|
// Footnote or endnote
|
|
@@ -263,7 +290,10 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
263
290
|
type: 'paragraph',
|
|
264
291
|
text: npContent.text,
|
|
265
292
|
children: npContent.children,
|
|
266
|
-
metadata:
|
|
293
|
+
metadata: {
|
|
294
|
+
...(npContent.alignment ? { alignment: npContent.alignment } : {}),
|
|
295
|
+
...(npContent.anchorIds?.length ? { anchorIds: npContent.anchorIds } : {})
|
|
296
|
+
}
|
|
267
297
|
};
|
|
268
298
|
noteChildren.push(npNode);
|
|
269
299
|
}
|
|
@@ -323,7 +353,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
323
353
|
}
|
|
324
354
|
}
|
|
325
355
|
}
|
|
326
|
-
return { text: fullText, children };
|
|
356
|
+
return { text: fullText, children, anchorIds };
|
|
327
357
|
};
|
|
328
358
|
/**
|
|
329
359
|
* Helper to parse a paragraph node (text:p or text:h) and extract its content.
|
|
@@ -394,7 +424,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
394
424
|
}
|
|
395
425
|
}
|
|
396
426
|
}
|
|
397
|
-
return { text: content.text, children: content.children, alignment, style: paraStyle || undefined };
|
|
427
|
+
return { text: content.text, children: content.children, alignment, style: paraStyle || undefined, anchorIds: content.anchorIds };
|
|
398
428
|
};
|
|
399
429
|
/**
|
|
400
430
|
* Splits paragraph content into multiple segments based on line breaks.
|
|
@@ -633,8 +663,9 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
633
663
|
(0, xmlUtils_js_1.getDirectChildren)(officeBody, "office:spreadsheet")[0];
|
|
634
664
|
if (bodyContent) {
|
|
635
665
|
const bodyChildren = (0, xmlUtils_js_1.getDirectChildren)(bodyContent, "*");
|
|
666
|
+
const isSpreadsheet = bodyContent.tagName === "office:spreadsheet";
|
|
636
667
|
for (const child of bodyChildren) {
|
|
637
|
-
traverse(child, content, false, xmlString);
|
|
668
|
+
traverse(child, content, false, xmlString, isSpreadsheet);
|
|
638
669
|
}
|
|
639
670
|
}
|
|
640
671
|
}
|
|
@@ -646,19 +677,28 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
646
677
|
* @param targetArray - The array to push extracted content nodes to
|
|
647
678
|
* @param forceHeading - If true, treats all paragraphs as headings (used for slide titles)
|
|
648
679
|
* @param sourceXml - The source XML string for raw content extraction
|
|
680
|
+
* @param asSheet - If true, treats tables as sheets (for ODS)
|
|
649
681
|
*/
|
|
650
|
-
function traverse(node, targetArray, forceHeading = false, sourceXml) {
|
|
682
|
+
function traverse(node, targetArray, forceHeading = false, sourceXml, asSheet = false) {
|
|
651
683
|
if (node.tagName === "text:p") {
|
|
652
684
|
const pContent = parseParagraphContent(node, paragraphStyleMap, styleMap, config, sourceXml);
|
|
653
685
|
const type = (forceHeading || (node.getAttribute("text:style-name") || '').toLowerCase().includes('title')) ? 'heading' : 'paragraph';
|
|
686
|
+
const metadata = {
|
|
687
|
+
...(pContent.alignment ? { alignment: pContent.alignment } : {}),
|
|
688
|
+
...(pContent.style ? { style: pContent.style } : {}),
|
|
689
|
+
...(pContent.anchorIds?.length ? { anchorIds: pContent.anchorIds } : {})
|
|
690
|
+
};
|
|
691
|
+
const nodeId = node.getAttribute("xml:id") || node.getAttribute("text:id");
|
|
692
|
+
if (nodeId) {
|
|
693
|
+
if (!metadata.anchorIds)
|
|
694
|
+
metadata.anchorIds = [];
|
|
695
|
+
metadata.anchorIds.push(nodeId);
|
|
696
|
+
}
|
|
654
697
|
const pNode = {
|
|
655
698
|
type,
|
|
656
699
|
text: pContent.text,
|
|
657
700
|
children: pContent.children,
|
|
658
|
-
metadata
|
|
659
|
-
...(pContent.alignment ? { alignment: pContent.alignment } : {}),
|
|
660
|
-
...(pContent.style ? { style: pContent.style } : {})
|
|
661
|
-
}
|
|
701
|
+
metadata
|
|
662
702
|
};
|
|
663
703
|
if (type === 'heading' && pNode.metadata) {
|
|
664
704
|
pNode.metadata.level = pNode.metadata.level || 1;
|
|
@@ -675,15 +715,23 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
675
715
|
else if (node.tagName === "text:h") {
|
|
676
716
|
const level = parseInt(node.getAttribute("text:outline-level") || "1");
|
|
677
717
|
const hContent = parseParagraphContent(node, paragraphStyleMap, styleMap, config, sourceXml);
|
|
718
|
+
const metadata = {
|
|
719
|
+
level,
|
|
720
|
+
...(hContent.alignment ? { alignment: hContent.alignment } : {}),
|
|
721
|
+
...(hContent.style ? { style: hContent.style } : {}),
|
|
722
|
+
...(hContent.anchorIds?.length ? { anchorIds: hContent.anchorIds } : {})
|
|
723
|
+
};
|
|
724
|
+
const nodeId = node.getAttribute("xml:id") || node.getAttribute("text:id");
|
|
725
|
+
if (nodeId) {
|
|
726
|
+
if (!metadata.anchorIds)
|
|
727
|
+
metadata.anchorIds = [];
|
|
728
|
+
metadata.anchorIds.push(nodeId);
|
|
729
|
+
}
|
|
678
730
|
const hNode = {
|
|
679
731
|
type: 'heading',
|
|
680
732
|
text: hContent.text,
|
|
681
733
|
children: hContent.children,
|
|
682
|
-
metadata
|
|
683
|
-
level,
|
|
684
|
-
...(hContent.alignment ? { alignment: hContent.alignment } : {}),
|
|
685
|
-
...(hContent.style ? { style: hContent.style } : {})
|
|
686
|
-
}
|
|
734
|
+
metadata
|
|
687
735
|
};
|
|
688
736
|
if (config.includeRawContent) {
|
|
689
737
|
hNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
|
|
@@ -694,6 +742,20 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
694
742
|
else if (node.tagName === "table:table") {
|
|
695
743
|
// Parse table with proper structure
|
|
696
744
|
const tableNode = parseTable(node, paragraphStyleMap, styleMap, config, sourceXml);
|
|
745
|
+
if (asSheet) {
|
|
746
|
+
tableNode.type = 'sheet';
|
|
747
|
+
const sheetName = node.getAttribute("table:name");
|
|
748
|
+
if (sheetName) {
|
|
749
|
+
tableNode.metadata = { ...tableNode.metadata, sheetName };
|
|
750
|
+
}
|
|
751
|
+
}
|
|
752
|
+
const tableId = node.getAttribute("xml:id") || node.getAttribute("table:name");
|
|
753
|
+
if (tableId) {
|
|
754
|
+
if (!tableNode.metadata)
|
|
755
|
+
tableNode.metadata = {};
|
|
756
|
+
tableNode.metadata.anchorIds = tableNode.metadata.anchorIds || [];
|
|
757
|
+
tableNode.metadata.anchorIds.push(tableId);
|
|
758
|
+
}
|
|
697
759
|
if (config.includeRawContent) {
|
|
698
760
|
tableNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
|
|
699
761
|
}
|
|
@@ -720,9 +782,8 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
720
782
|
parentNode = parentNode.parentNode;
|
|
721
783
|
}
|
|
722
784
|
}
|
|
723
|
-
// Try to find list style in automatic styles to determine type and visibility
|
|
785
|
+
// Try to find list style in automatic styles or styles.xml to determine type and visibility
|
|
724
786
|
if (styleNameToCheck) {
|
|
725
|
-
const automaticStyles = (0, xmlUtils_js_1.getFirstElementByTagName)((0, xmlUtils_js_1.parseXmlString)(mainContentFile?.content.toString() || ''), "office:automatic-styles");
|
|
726
787
|
if (automaticStyles) {
|
|
727
788
|
const listStyles = (0, xmlUtils_js_1.getElementsByTagName)(automaticStyles, "text:list-style");
|
|
728
789
|
for (const listStyle of listStyles) {
|
|
@@ -739,32 +800,32 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
739
800
|
listType = 'unordered';
|
|
740
801
|
isVisible = bulletLevels.some(l => !!l.getAttribute("text:bullet-char"));
|
|
741
802
|
}
|
|
742
|
-
if (imageLevels.length > 0)
|
|
803
|
+
else if (imageLevels.length > 0) {
|
|
804
|
+
listType = 'unordered';
|
|
743
805
|
isVisible = true;
|
|
806
|
+
}
|
|
744
807
|
break;
|
|
745
808
|
}
|
|
746
809
|
}
|
|
747
810
|
}
|
|
748
|
-
|
|
749
|
-
|
|
750
|
-
|
|
751
|
-
|
|
752
|
-
|
|
753
|
-
|
|
754
|
-
|
|
755
|
-
|
|
756
|
-
|
|
757
|
-
|
|
758
|
-
|
|
759
|
-
|
|
760
|
-
|
|
761
|
-
|
|
762
|
-
|
|
763
|
-
|
|
811
|
+
if (!isVisible && stylesDom) {
|
|
812
|
+
const officeStyles = (0, xmlUtils_js_1.getFirstElementByTagName)(stylesDom, "office:styles");
|
|
813
|
+
if (officeStyles) {
|
|
814
|
+
const listStyles = (0, xmlUtils_js_1.getElementsByTagName)(officeStyles, "text:list-style");
|
|
815
|
+
for (const listStyle of listStyles) {
|
|
816
|
+
if (listStyle.getAttribute("style:name") === styleNameToCheck) {
|
|
817
|
+
const bulletLevels = (0, xmlUtils_js_1.getElementsByTagName)(listStyle, "text:list-level-style-bullet");
|
|
818
|
+
const numberLevels = (0, xmlUtils_js_1.getElementsByTagName)(listStyle, "text:list-level-style-number");
|
|
819
|
+
if (numberLevels.length > 0) {
|
|
820
|
+
listType = 'ordered';
|
|
821
|
+
isVisible = numberLevels.some(l => !!l.getAttribute("style:num-format"));
|
|
822
|
+
}
|
|
823
|
+
else if (bulletLevels.length > 0) {
|
|
824
|
+
listType = 'unordered';
|
|
825
|
+
isVisible = bulletLevels.some(l => !!l.getAttribute("text:bullet-char"));
|
|
826
|
+
}
|
|
827
|
+
break;
|
|
764
828
|
}
|
|
765
|
-
if (imageLevels.length > 0)
|
|
766
|
-
isVisible = true;
|
|
767
|
-
break;
|
|
768
829
|
}
|
|
769
830
|
}
|
|
770
831
|
}
|
|
@@ -919,14 +980,19 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
919
980
|
const parts = imageHref.split('/');
|
|
920
981
|
imageHref = parts[parts.length - 1];
|
|
921
982
|
}
|
|
983
|
+
const metadata = {
|
|
984
|
+
attachmentName: imageHref,
|
|
985
|
+
...(altText ? { altText } : {})
|
|
986
|
+
};
|
|
987
|
+
const frameId = node.getAttribute("xml:id") || node.getAttribute("draw:name");
|
|
988
|
+
if (frameId) {
|
|
989
|
+
metadata.anchorIds = [frameId];
|
|
990
|
+
}
|
|
922
991
|
const imageNode = {
|
|
923
992
|
type: 'image',
|
|
924
993
|
text: '',
|
|
925
994
|
children: [],
|
|
926
|
-
metadata
|
|
927
|
-
attachmentName: imageHref,
|
|
928
|
-
...(altText ? { altText } : {})
|
|
929
|
-
}
|
|
995
|
+
metadata
|
|
930
996
|
};
|
|
931
997
|
if (config.includeRawContent) {
|
|
932
998
|
imageNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
|
|
@@ -1259,10 +1325,10 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1259
1325
|
if (config.ocr) {
|
|
1260
1326
|
if (attachment.mimeType.startsWith('image/')) {
|
|
1261
1327
|
try {
|
|
1262
|
-
attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(media.content, {
|
|
1328
|
+
attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(media.content, { ...config.ocrConfig })).trim();
|
|
1263
1329
|
}
|
|
1264
1330
|
catch (e) {
|
|
1265
|
-
(0, errorUtils_js_1.logWarning)(
|
|
1331
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.OCR_FAILED, config, attachment.name, e);
|
|
1266
1332
|
}
|
|
1267
1333
|
}
|
|
1268
1334
|
}
|
|
@@ -1399,7 +1465,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1399
1465
|
node.text = attachment.ocrText;
|
|
1400
1466
|
}
|
|
1401
1467
|
if (attachment.chartData && node.type === 'chart') {
|
|
1402
|
-
node.text = attachment.chartData.rawTexts.join(config.newlineDelimiter
|
|
1468
|
+
node.text = attachment.chartData.rawTexts.join(config.newlineDelimiter);
|
|
1403
1469
|
}
|
|
1404
1470
|
}
|
|
1405
1471
|
}
|
|
@@ -1433,32 +1499,27 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1433
1499
|
if (config.putNotesAtLast && notes.length > 0) {
|
|
1434
1500
|
content.push(...notes);
|
|
1435
1501
|
}
|
|
1436
|
-
|
|
1437
|
-
|
|
1438
|
-
|
|
1439
|
-
|
|
1440
|
-
|
|
1441
|
-
|
|
1442
|
-
|
|
1443
|
-
|
|
1444
|
-
|
|
1445
|
-
|
|
1446
|
-
|
|
1447
|
-
|
|
1448
|
-
|
|
1449
|
-
|
|
1450
|
-
|
|
1451
|
-
|
|
1452
|
-
|
|
1453
|
-
|
|
1454
|
-
|
|
1455
|
-
|
|
1456
|
-
|
|
1457
|
-
|
|
1458
|
-
return t;
|
|
1459
|
-
};
|
|
1460
|
-
return getText(c);
|
|
1461
|
-
}).filter(t => t != '').join(config.newlineDelimiter ?? '\n')
|
|
1462
|
-
};
|
|
1502
|
+
const toTextSync = () => content.map(c => {
|
|
1503
|
+
const getText = (node) => {
|
|
1504
|
+
let t = '';
|
|
1505
|
+
if (node.children && node.children.length > 0) {
|
|
1506
|
+
// Check if children have their own children (container vs leaf)
|
|
1507
|
+
// If children are leaf nodes (text/image), join with empty string
|
|
1508
|
+
// If children are container nodes (paragraphs/rows), join with newline
|
|
1509
|
+
const hasGrandChildren = node.children.some(child => child.children && child.children.length > 0);
|
|
1510
|
+
const separator = hasGrandChildren ? config.newlineDelimiter : '';
|
|
1511
|
+
t += node.children.map(getText).filter(t => t != '').join(separator);
|
|
1512
|
+
}
|
|
1513
|
+
else {
|
|
1514
|
+
t += node.text || '';
|
|
1515
|
+
}
|
|
1516
|
+
return t;
|
|
1517
|
+
};
|
|
1518
|
+
return getText(c);
|
|
1519
|
+
}).filter(t => t != '').join(config.newlineDelimiter);
|
|
1520
|
+
return (0, astUtils_js_1.createAST)(fileType, {
|
|
1521
|
+
...metadata,
|
|
1522
|
+
styleMap: combinedStyleMap
|
|
1523
|
+
}, content, attachments, config, toTextSync);
|
|
1463
1524
|
};
|
|
1464
1525
|
exports.parseOpenOffice = parseOpenOffice;
|
|
@@ -56,7 +56,7 @@
|
|
|
56
56
|
* @see https://mozilla.github.io/pdf.js/ PDF.js documentation
|
|
57
57
|
* @see https://www.adobe.com/devnet/pdf/pdf_reference.html PDF Reference
|
|
58
58
|
*/
|
|
59
|
-
import {
|
|
59
|
+
import { FullOfficeParserConfig, OfficeParserAST } from '../types.js';
|
|
60
60
|
/**
|
|
61
61
|
* Parses a PDF file and extracts content.
|
|
62
62
|
*
|
|
@@ -64,4 +64,4 @@ import { OfficeParserAST, OfficeParserConfig } from '../types.js';
|
|
|
64
64
|
* @param config - Parser configuration
|
|
65
65
|
* @returns Promise resolving to the parsed AST
|
|
66
66
|
*/
|
|
67
|
-
export declare const parsePdf: (buffer: Buffer, config:
|
|
67
|
+
export declare const parsePdf: (buffer: Buffer, config: FullOfficeParserConfig) => Promise<OfficeParserAST>;
|
|
@@ -59,12 +59,15 @@
|
|
|
59
59
|
*/
|
|
60
60
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
61
61
|
exports.parsePdf = void 0;
|
|
62
|
+
const defaults_js_1 = require("../defaults.js");
|
|
63
|
+
const types_js_1 = require("../types.js");
|
|
64
|
+
const astUtils_js_1 = require("../utils/astUtils.js");
|
|
65
|
+
const dateUtils_js_1 = require("../utils/dateUtils.js");
|
|
66
|
+
const envUtils_js_1 = require("../utils/envUtils.js");
|
|
62
67
|
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
63
68
|
const imageUtils_js_1 = require("../utils/imageUtils.js");
|
|
64
|
-
const ocrUtils_js_1 = require("../utils/ocrUtils.js");
|
|
65
69
|
const moduleLoader_js_1 = require("../utils/moduleLoader.js");
|
|
66
|
-
const
|
|
67
|
-
const envUtils_js_1 = require("../utils/envUtils.js");
|
|
70
|
+
const ocrUtils_js_1 = require("../utils/ocrUtils.js");
|
|
68
71
|
/** Type guard for TextItem in PDF.js 5.x */
|
|
69
72
|
function isTextItem(item) {
|
|
70
73
|
return item && typeof item.str === 'string' && Array.isArray(item.transform) && item.transform.length >= 6;
|
|
@@ -265,29 +268,38 @@ function convertToRgbaBuffer(data, width, height, kind) {
|
|
|
265
268
|
const parsePdf = async (buffer, config) => {
|
|
266
269
|
const pdfjs = await (0, moduleLoader_js_1.loadPdfJs)();
|
|
267
270
|
// Configure worker
|
|
268
|
-
|
|
269
|
-
|
|
271
|
+
const workerSrc = config.pdfWorkerSrc;
|
|
272
|
+
if (envUtils_js_1.isBrowser) {
|
|
273
|
+
pdfjs.GlobalWorkerOptions.workerSrc = workerSrc;
|
|
270
274
|
}
|
|
271
275
|
else {
|
|
272
|
-
//
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
+
// Node.js: Try to auto-resolve local worker path to avoid remote fetch errors
|
|
277
|
+
(0, envUtils_js_1.assertNode)('pdf-worker-auto-resolution');
|
|
278
|
+
let resolved = false;
|
|
279
|
+
// If the user provided a custom path (not the default CDN one), use it.
|
|
280
|
+
// Otherwise, try to find it locally.
|
|
281
|
+
if (workerSrc !== defaults_js_1.DEFAULT_OFFICE_PARSER_CONFIG.pdfWorkerSrc && workerSrc !== '') {
|
|
282
|
+
pdfjs.GlobalWorkerOptions.workerSrc = workerSrc;
|
|
283
|
+
resolved = true;
|
|
276
284
|
}
|
|
277
285
|
else {
|
|
278
|
-
// Node.js: Try to auto-resolve local worker path to avoid remote fetch errors
|
|
279
|
-
(0, envUtils_js_1.assertNode)('pdf-worker-auto-resolution');
|
|
280
286
|
try {
|
|
281
287
|
// We use require.resolve to find the exact path of the installed package.
|
|
282
288
|
// @ts-ignore - 'require' is available in Node.js/CommonJS environment
|
|
283
|
-
const
|
|
284
|
-
|
|
289
|
+
const localWorkerPath = require.resolve('pdfjs-dist/legacy/build/pdf.worker.mjs');
|
|
290
|
+
// Use file:// URL for the worker source in Node.js to ensure compatibility with ESM-native PDF.js 5.x
|
|
291
|
+
// We use dynamic import for 'url' to avoid breaking browser bundles
|
|
292
|
+
const { pathToFileURL } = await import('url');
|
|
293
|
+
pdfjs.GlobalWorkerOptions.workerSrc = pathToFileURL(localWorkerPath).href;
|
|
294
|
+
resolved = true;
|
|
285
295
|
}
|
|
286
296
|
catch (e) {
|
|
287
|
-
|
|
288
|
-
console.warn("[PdfParser] Could not auto-resolve local worker path:", e);
|
|
297
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.PDF_WORKER_FALLBACK, config, undefined, e);
|
|
289
298
|
}
|
|
290
299
|
}
|
|
300
|
+
if (!resolved) {
|
|
301
|
+
pdfjs.GlobalWorkerOptions.workerSrc = workerSrc;
|
|
302
|
+
}
|
|
291
303
|
}
|
|
292
304
|
const uint8Array = new Uint8Array(buffer);
|
|
293
305
|
const loadingTask = pdfjs.getDocument({
|
|
@@ -302,7 +314,7 @@ const parsePdf = async (buffer, config) => {
|
|
|
302
314
|
catch (e) {
|
|
303
315
|
const message = e instanceof Error ? e.message : String(e);
|
|
304
316
|
if (message.includes('workerSrc') || message.includes('No "GlobalWorkerOptions.workerSrc" specified')) {
|
|
305
|
-
throw (0, errorUtils_js_1.getOfficeError)(
|
|
317
|
+
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.PDF_WORKER_MISSING, config);
|
|
306
318
|
}
|
|
307
319
|
throw e;
|
|
308
320
|
}
|
|
@@ -382,8 +394,7 @@ const parsePdf = async (buffer, config) => {
|
|
|
382
394
|
}
|
|
383
395
|
}
|
|
384
396
|
catch (e) {
|
|
385
|
-
|
|
386
|
-
console.error("Error extracting embedded attachments:", e);
|
|
397
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.ATTACHMENT_EXTRACTION_FAILED, config, undefined, e);
|
|
387
398
|
}
|
|
388
399
|
// --- First Pass: Collect all items for font statistics ---
|
|
389
400
|
for (let i = 1; i <= numPages; i++) {
|
|
@@ -396,13 +407,13 @@ const parsePdf = async (buffer, config) => {
|
|
|
396
407
|
textContent = await page.getTextContent();
|
|
397
408
|
}
|
|
398
409
|
catch (e) {
|
|
399
|
-
|
|
400
|
-
console.warn(`[PdfParser] Error loading page ${i}:`, e);
|
|
410
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.PAGE_LOAD_FAILED, config, i, e);
|
|
401
411
|
// Push empty items to maintain index alignment for second pass
|
|
402
412
|
allPageItems.push(pageItems);
|
|
403
413
|
continue;
|
|
404
414
|
}
|
|
405
415
|
const commonObjs = page.commonObjs;
|
|
416
|
+
const fontCache = new Map();
|
|
406
417
|
for (const item of textContent.items) {
|
|
407
418
|
// PDF.js 5.x: textContent.items can contain TextMarkedContent which lack
|
|
408
419
|
// 'str' and 'transform'. Skip these to avoid crashes and page skipping.
|
|
@@ -422,11 +433,15 @@ const parsePdf = async (buffer, config) => {
|
|
|
422
433
|
if (textItem.fontName && commonObjs) {
|
|
423
434
|
try {
|
|
424
435
|
if (commonObjs.has(textItem.fontName)) {
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
//
|
|
428
|
-
|
|
429
|
-
|
|
436
|
+
let fontData = fontCache.get(textItem.fontName);
|
|
437
|
+
if (!fontData) {
|
|
438
|
+
// Use callback-based get to ensure safe resolution
|
|
439
|
+
fontData = await new Promise((resolve) => {
|
|
440
|
+
// @ts-ignore - commonObjs.get is callback-based in legacy builds
|
|
441
|
+
commonObjs.get(textItem.fontName, (data) => resolve(data));
|
|
442
|
+
});
|
|
443
|
+
fontCache.set(textItem.fontName, fontData);
|
|
444
|
+
}
|
|
430
445
|
if (fontData?.name && typeof fontData.name === 'string') {
|
|
431
446
|
// Remove PDF subset prefix (6 uppercase letters + '+')
|
|
432
447
|
fontName = fontData.name.replace(/^[A-Z]{6}\+/, '');
|
|
@@ -485,9 +500,7 @@ const parsePdf = async (buffer, config) => {
|
|
|
485
500
|
});
|
|
486
501
|
}
|
|
487
502
|
catch (e) {
|
|
488
|
-
|
|
489
|
-
console.error(`[PdfParser] Failed to load dependency ${dep}:`, e);
|
|
490
|
-
}
|
|
503
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.DEPENDENCY_LOAD_FAILED, config, dep, e);
|
|
491
504
|
}
|
|
492
505
|
}
|
|
493
506
|
}
|
|
@@ -507,7 +520,7 @@ const parsePdf = async (buffer, config) => {
|
|
|
507
520
|
targetObjs.get(imgName, (data) => resolve(data));
|
|
508
521
|
});
|
|
509
522
|
// Browser-specific: Handle ImageBitmap if data is missing
|
|
510
|
-
if (
|
|
523
|
+
if (envUtils_js_1.isBrowser && !imgObj.data && imgObj.bitmap) {
|
|
511
524
|
try {
|
|
512
525
|
const canvas = document.createElement('canvas');
|
|
513
526
|
canvas.width = imgObj.width;
|
|
@@ -520,8 +533,7 @@ const parsePdf = async (buffer, config) => {
|
|
|
520
533
|
}
|
|
521
534
|
}
|
|
522
535
|
catch (e) {
|
|
523
|
-
|
|
524
|
-
console.error(`[PdfParser] Failed to extract from ImageBitmap:`, e);
|
|
536
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.IMAGE_PROCESSING_FAILED, config, undefined, e);
|
|
525
537
|
}
|
|
526
538
|
}
|
|
527
539
|
if (imgObj?.data && imgObj.width > 0 && imgObj.height > 0) {
|
|
@@ -554,8 +566,7 @@ const parsePdf = async (buffer, config) => {
|
|
|
554
566
|
}
|
|
555
567
|
}
|
|
556
568
|
catch (e) {
|
|
557
|
-
|
|
558
|
-
console.error(`Error extracting images from page ${i}:`, e);
|
|
569
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.IMAGE_EXTRACTION_FAILED, config, `from page ${i}`, e);
|
|
559
570
|
}
|
|
560
571
|
}
|
|
561
572
|
allPageItems.push(pageItems);
|
|
@@ -570,8 +581,7 @@ const parsePdf = async (buffer, config) => {
|
|
|
570
581
|
page = await pdfDocument.getPage(pageNum);
|
|
571
582
|
}
|
|
572
583
|
catch (e) {
|
|
573
|
-
|
|
574
|
-
console.warn(`[PdfParser] Error loading page ${pageNum} in second pass:`, e);
|
|
584
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.PAGE_LOAD_FAILED, config, pageNum, e);
|
|
575
585
|
continue;
|
|
576
586
|
}
|
|
577
587
|
const pageItems = allPageItems[i];
|
|
@@ -594,8 +604,7 @@ const parsePdf = async (buffer, config) => {
|
|
|
594
604
|
}
|
|
595
605
|
}
|
|
596
606
|
catch (e) {
|
|
597
|
-
|
|
598
|
-
console.error(`Error extracting annotations from page ${pageNum}:`, e);
|
|
607
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.ANNOTATION_EXTRACTION_FAILED, config, pageNum, e);
|
|
599
608
|
}
|
|
600
609
|
// Sort items: Y descending (top to bottom), then X ascending (left to right)
|
|
601
610
|
pageItems.sort((a, b) => {
|
|
@@ -744,12 +753,11 @@ const parsePdf = async (buffer, config) => {
|
|
|
744
753
|
try {
|
|
745
754
|
// Skip OCR for very small images/artifacts (e.g. < 10px) to avoid Tesseract warnings
|
|
746
755
|
if (item.width >= 10 && item.height >= 10) {
|
|
747
|
-
attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(bmpBuffer, {
|
|
756
|
+
attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(bmpBuffer, { ...config.ocrConfig })).trim();
|
|
748
757
|
}
|
|
749
758
|
}
|
|
750
759
|
catch (e) {
|
|
751
|
-
|
|
752
|
-
console.error(`OCR failed for ${attachmentName}:`, e);
|
|
760
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.OCR_FAILED, config, attachmentName, e);
|
|
753
761
|
}
|
|
754
762
|
}
|
|
755
763
|
attachments.push(attachment);
|
|
@@ -764,7 +772,7 @@ const parsePdf = async (buffer, config) => {
|
|
|
764
772
|
});
|
|
765
773
|
}
|
|
766
774
|
catch (e) {
|
|
767
|
-
(0, errorUtils_js_1.logWarning)(
|
|
775
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.IMAGE_EXTRACTION_FAILED, config, attachmentName, e);
|
|
768
776
|
}
|
|
769
777
|
}
|
|
770
778
|
}
|
|
@@ -777,17 +785,12 @@ const parsePdf = async (buffer, config) => {
|
|
|
777
785
|
content.push({
|
|
778
786
|
type: 'page',
|
|
779
787
|
children: pageContent,
|
|
780
|
-
text: pageContent.map(node => node.text).join(config.newlineDelimiter
|
|
788
|
+
text: pageContent.map(node => node.text).join(config.newlineDelimiter),
|
|
781
789
|
metadata: { pageNumber: pageNum }
|
|
782
790
|
});
|
|
783
791
|
}
|
|
784
|
-
|
|
785
|
-
|
|
786
|
-
metadata: metadata,
|
|
787
|
-
content: content,
|
|
788
|
-
attachments: attachments,
|
|
789
|
-
toText: () => content.map(c => c.text).join(config.newlineDelimiter ?? '\n\n')
|
|
790
|
-
};
|
|
792
|
+
const toTextSync = () => content.map(c => c.text).join(config.newlineDelimiter);
|
|
793
|
+
return (0, astUtils_js_1.createAST)('pdf', metadata, content, attachments, config, toTextSync);
|
|
791
794
|
};
|
|
792
795
|
exports.parsePdf = parsePdf;
|
|
793
796
|
/**
|
|
@@ -21,7 +21,7 @@
|
|
|
21
21
|
* @module PowerPointParser
|
|
22
22
|
* @see https://www.ecma-international.org/publications-and-standards/standards/ecma-376/
|
|
23
23
|
*/
|
|
24
|
-
import {
|
|
24
|
+
import { FullOfficeParserConfig, OfficeParserAST } from '../types.js';
|
|
25
25
|
/**
|
|
26
26
|
* Parses a PowerPoint presentation (.pptx) and extracts slides and notes.
|
|
27
27
|
*
|
|
@@ -29,4 +29,4 @@ import { OfficeParserAST, OfficeParserConfig } from '../types.js';
|
|
|
29
29
|
* @param config - Parser configuration
|
|
30
30
|
* @returns A promise resolving to the parsed AST
|
|
31
31
|
*/
|
|
32
|
-
export declare const parsePowerPoint: (buffer: Buffer, config:
|
|
32
|
+
export declare const parsePowerPoint: (buffer: Buffer, config: FullOfficeParserConfig) => Promise<OfficeParserAST>;
|