officeparser 6.1.1 → 7.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (70) hide show
  1. package/README.md +219 -26
  2. package/dist/OfficeConverter.d.ts +46 -0
  3. package/dist/OfficeConverter.js +72 -0
  4. package/dist/OfficeGenerator.d.ts +19 -0
  5. package/dist/OfficeGenerator.js +48 -0
  6. package/dist/OfficeParser.d.ts +6 -0
  7. package/dist/OfficeParser.js +55 -29
  8. package/dist/cli.d.ts +3 -1
  9. package/dist/cli.js +106 -22
  10. package/dist/defaults.d.ts +41 -0
  11. package/dist/defaults.js +172 -0
  12. package/dist/generators/BaseGenerator.d.ts +58 -0
  13. package/dist/generators/BaseGenerator.js +107 -0
  14. package/dist/generators/ChunkingGenerator.d.ts +81 -0
  15. package/dist/generators/ChunkingGenerator.js +683 -0
  16. package/dist/generators/CsvGenerator.d.ts +30 -0
  17. package/dist/generators/CsvGenerator.js +233 -0
  18. package/dist/generators/HtmlGenerator.d.ts +37 -0
  19. package/dist/generators/HtmlGenerator.js +1013 -0
  20. package/dist/generators/MarkdownGenerator.d.ts +59 -0
  21. package/dist/generators/MarkdownGenerator.js +481 -0
  22. package/dist/generators/PdfGenerator.d.ts +22 -0
  23. package/dist/generators/PdfGenerator.js +118 -0
  24. package/dist/generators/RtfGenerator.d.ts +15 -0
  25. package/dist/generators/RtfGenerator.js +208 -0
  26. package/dist/generators/TextGenerator.d.ts +13 -0
  27. package/dist/generators/TextGenerator.js +108 -0
  28. package/dist/index.d.ts +11 -3
  29. package/dist/index.js +17 -2
  30. package/dist/index.mjs +2 -2
  31. package/dist/officeparser.browser.d.ts +826 -5
  32. package/dist/officeparser.browser.iife.js +703 -52
  33. package/dist/officeparser.browser.mjs +703 -52
  34. package/dist/parsers/CsvParser.d.ts +9 -0
  35. package/dist/parsers/CsvParser.js +110 -0
  36. package/dist/parsers/ExcelParser.d.ts +2 -2
  37. package/dist/parsers/ExcelParser.js +145 -114
  38. package/dist/parsers/HtmlParser.d.ts +2 -0
  39. package/dist/parsers/HtmlParser.js +539 -0
  40. package/dist/parsers/MarkdownParser.d.ts +2 -0
  41. package/dist/parsers/MarkdownParser.js +360 -0
  42. package/dist/parsers/OpenOfficeParser.d.ts +2 -2
  43. package/dist/parsers/OpenOfficeParser.js +140 -79
  44. package/dist/parsers/PdfParser.d.ts +2 -2
  45. package/dist/parsers/PdfParser.js +52 -49
  46. package/dist/parsers/PowerPointParser.d.ts +2 -2
  47. package/dist/parsers/PowerPointParser.js +20 -23
  48. package/dist/parsers/RtfParser.d.ts +2 -2
  49. package/dist/parsers/RtfParser.js +1291 -1240
  50. package/dist/parsers/WordParser.d.ts +2 -2
  51. package/dist/parsers/WordParser.js +232 -97
  52. package/dist/sbom.cdx.json +99 -99
  53. package/dist/types.d.ts +781 -5
  54. package/dist/types.js +71 -0
  55. package/dist/utils/astUtils.d.ts +16 -0
  56. package/dist/utils/astUtils.js +32 -0
  57. package/dist/utils/configUtils.d.ts +26 -0
  58. package/dist/utils/configUtils.js +140 -0
  59. package/dist/utils/envUtils.js +56 -2
  60. package/dist/utils/errorUtils.d.ts +17 -29
  61. package/dist/utils/errorUtils.js +109 -52
  62. package/dist/utils/moduleLoader.js +15 -9
  63. package/dist/utils/ocrUtils.js +2 -1
  64. package/dist/utils/sheetUtils.d.ts +7 -0
  65. package/dist/utils/sheetUtils.js +35 -0
  66. package/dist/utils/styleMapper.d.ts +36 -0
  67. package/dist/utils/styleMapper.js +224 -0
  68. package/dist/utils/xmlUtils.d.ts +0 -8
  69. package/dist/utils/xmlUtils.js +2 -1
  70. package/package.json +27 -8
@@ -23,6 +23,8 @@
23
23
  */
24
24
  Object.defineProperty(exports, "__esModule", { value: true });
25
25
  exports.parseOpenOffice = void 0;
26
+ const types_js_1 = require("../types.js");
27
+ const astUtils_js_1 = require("../utils/astUtils.js");
26
28
  const chartUtils_js_1 = require("../utils/chartUtils.js");
27
29
  const errorUtils_js_1 = require("../utils/errorUtils.js");
28
30
  const imageUtils_js_1 = require("../utils/imageUtils.js");
@@ -63,6 +65,7 @@ const parseOpenOffice = async (buffer, config) => {
63
65
  }
64
66
  const mainContentFile = files.find(f => f.path === 'content.xml') || files.find(f => f.path.match(contentFileRegex));
65
67
  const stylesFile = files.find(f => f.path === 'styles.xml');
68
+ const stylesDom = stylesFile ? (0, xmlUtils_js_1.parseXmlString)(stylesFile.content.toString()) : undefined;
66
69
  const content = [];
67
70
  const notes = [];
68
71
  // Style Map: styleName -> TextFormatting
@@ -76,8 +79,7 @@ const parseOpenOffice = async (buffer, config) => {
76
79
  let listIdCounter = 0;
77
80
  let lastWasList = false;
78
81
  // Helper to parse styles
79
- const parseStyles = (xmlString) => {
80
- const xml = (0, xmlUtils_js_1.parseXmlString)(xmlString);
82
+ const parseStyles = (xml) => {
81
83
  const styles = (0, xmlUtils_js_1.getElementsByTagName)(xml, "style:style");
82
84
  for (const style of styles) {
83
85
  const name = style.getAttribute("style:name");
@@ -156,8 +158,8 @@ const parseOpenOffice = async (buffer, config) => {
156
158
  }
157
159
  }
158
160
  };
159
- if (stylesFile) {
160
- parseStyles(stylesFile.content.toString());
161
+ if (stylesDom) {
162
+ parseStyles(stylesDom);
161
163
  }
162
164
  /**
163
165
  * Helper to parse a paragraph node (text:p or text:h) and extract its content.
@@ -177,9 +179,10 @@ const parseOpenOffice = async (buffer, config) => {
177
179
  */
178
180
  const parseInlineContent = (node, styleMap, config, notes, paragraphStyleMap, parentFormatting = {}, linkMetadata, sourceXml = '') => {
179
181
  const children = [];
182
+ const anchorIds = [];
180
183
  let fullText = '';
181
184
  if (!node.childNodes)
182
- return { text: '', children: [] };
185
+ return { text: '', children: [], anchorIds: [] };
183
186
  for (let i = 0; i < node.childNodes.length; i++) {
184
187
  const child = node.childNodes[i];
185
188
  if (child.nodeType === 3) { // Text node
@@ -197,7 +200,12 @@ const parseOpenOffice = async (buffer, config) => {
197
200
  else if ((0, xmlUtils_js_1.isElement)(child)) {
198
201
  const element = child;
199
202
  const tagName = element.tagName;
200
- if (tagName === 'text:s') {
203
+ if (tagName === 'text:bookmark' || tagName === 'text:bookmark-start') {
204
+ const name = element.getAttribute('text:name');
205
+ if (name)
206
+ anchorIds.push(name);
207
+ }
208
+ else if (tagName === 'text:s') {
201
209
  // Space
202
210
  const count = parseInt(element.getAttribute('text:c') || '1');
203
211
  const spaces = ' '.repeat(count);
@@ -236,15 +244,34 @@ const parseOpenOffice = async (buffer, config) => {
236
244
  const spanContent = parseInlineContent(element, styleMap, config, notes, paragraphStyleMap, formatting, linkMetadata, sourceXml);
237
245
  fullText += spanContent.text;
238
246
  children.push(...spanContent.children);
247
+ anchorIds.push(...spanContent.anchorIds);
239
248
  }
240
249
  else if (tagName === 'text:a') {
241
250
  // Hyperlink
242
- const href = element.getAttribute('xlink:href') || '';
243
- const linkType = href.startsWith('#') ? 'internal' : 'external';
244
- const newLinkMetadata = { link: href, linkType: linkType };
251
+ let href = element.getAttribute('xlink:href') || '';
252
+ const isInternal = href.startsWith('#');
253
+ const linkType = isInternal ? 'internal' : 'external';
254
+ if (isInternal) {
255
+ // ODT internal links can be encoded and might have suffixes like |outline
256
+ try {
257
+ href = decodeURIComponent(href).split('|')[0];
258
+ }
259
+ catch (e) {
260
+ href = href.split('|')[0];
261
+ }
262
+ // Normalize internal link: if it contains #, keep only from # onwards
263
+ if (href.includes('#')) {
264
+ href = '#' + href.split('#').pop();
265
+ }
266
+ }
267
+ let newLinkMetadata;
268
+ if (!isInternal || !config.ignoreInternalLinks) {
269
+ newLinkMetadata = { link: href, linkType: linkType };
270
+ }
245
271
  const linkContent = parseInlineContent(element, styleMap, config, notes, paragraphStyleMap, parentFormatting, newLinkMetadata, sourceXml);
246
272
  fullText += linkContent.text;
247
273
  children.push(...linkContent.children);
274
+ anchorIds.push(...linkContent.anchorIds);
248
275
  }
249
276
  else if (tagName === 'text:note' && !config.ignoreNotes) {
250
277
  // Footnote or endnote
@@ -263,7 +290,10 @@ const parseOpenOffice = async (buffer, config) => {
263
290
  type: 'paragraph',
264
291
  text: npContent.text,
265
292
  children: npContent.children,
266
- metadata: npContent.alignment ? { alignment: npContent.alignment } : undefined
293
+ metadata: {
294
+ ...(npContent.alignment ? { alignment: npContent.alignment } : {}),
295
+ ...(npContent.anchorIds?.length ? { anchorIds: npContent.anchorIds } : {})
296
+ }
267
297
  };
268
298
  noteChildren.push(npNode);
269
299
  }
@@ -323,7 +353,7 @@ const parseOpenOffice = async (buffer, config) => {
323
353
  }
324
354
  }
325
355
  }
326
- return { text: fullText, children };
356
+ return { text: fullText, children, anchorIds };
327
357
  };
328
358
  /**
329
359
  * Helper to parse a paragraph node (text:p or text:h) and extract its content.
@@ -394,7 +424,7 @@ const parseOpenOffice = async (buffer, config) => {
394
424
  }
395
425
  }
396
426
  }
397
- return { text: content.text, children: content.children, alignment, style: paraStyle || undefined };
427
+ return { text: content.text, children: content.children, alignment, style: paraStyle || undefined, anchorIds: content.anchorIds };
398
428
  };
399
429
  /**
400
430
  * Splits paragraph content into multiple segments based on line breaks.
@@ -633,8 +663,9 @@ const parseOpenOffice = async (buffer, config) => {
633
663
  (0, xmlUtils_js_1.getDirectChildren)(officeBody, "office:spreadsheet")[0];
634
664
  if (bodyContent) {
635
665
  const bodyChildren = (0, xmlUtils_js_1.getDirectChildren)(bodyContent, "*");
666
+ const isSpreadsheet = bodyContent.tagName === "office:spreadsheet";
636
667
  for (const child of bodyChildren) {
637
- traverse(child, content, false, xmlString);
668
+ traverse(child, content, false, xmlString, isSpreadsheet);
638
669
  }
639
670
  }
640
671
  }
@@ -646,19 +677,28 @@ const parseOpenOffice = async (buffer, config) => {
646
677
  * @param targetArray - The array to push extracted content nodes to
647
678
  * @param forceHeading - If true, treats all paragraphs as headings (used for slide titles)
648
679
  * @param sourceXml - The source XML string for raw content extraction
680
+ * @param asSheet - If true, treats tables as sheets (for ODS)
649
681
  */
650
- function traverse(node, targetArray, forceHeading = false, sourceXml) {
682
+ function traverse(node, targetArray, forceHeading = false, sourceXml, asSheet = false) {
651
683
  if (node.tagName === "text:p") {
652
684
  const pContent = parseParagraphContent(node, paragraphStyleMap, styleMap, config, sourceXml);
653
685
  const type = (forceHeading || (node.getAttribute("text:style-name") || '').toLowerCase().includes('title')) ? 'heading' : 'paragraph';
686
+ const metadata = {
687
+ ...(pContent.alignment ? { alignment: pContent.alignment } : {}),
688
+ ...(pContent.style ? { style: pContent.style } : {}),
689
+ ...(pContent.anchorIds?.length ? { anchorIds: pContent.anchorIds } : {})
690
+ };
691
+ const nodeId = node.getAttribute("xml:id") || node.getAttribute("text:id");
692
+ if (nodeId) {
693
+ if (!metadata.anchorIds)
694
+ metadata.anchorIds = [];
695
+ metadata.anchorIds.push(nodeId);
696
+ }
654
697
  const pNode = {
655
698
  type,
656
699
  text: pContent.text,
657
700
  children: pContent.children,
658
- metadata: {
659
- ...(pContent.alignment ? { alignment: pContent.alignment } : {}),
660
- ...(pContent.style ? { style: pContent.style } : {})
661
- }
701
+ metadata
662
702
  };
663
703
  if (type === 'heading' && pNode.metadata) {
664
704
  pNode.metadata.level = pNode.metadata.level || 1;
@@ -675,15 +715,23 @@ const parseOpenOffice = async (buffer, config) => {
675
715
  else if (node.tagName === "text:h") {
676
716
  const level = parseInt(node.getAttribute("text:outline-level") || "1");
677
717
  const hContent = parseParagraphContent(node, paragraphStyleMap, styleMap, config, sourceXml);
718
+ const metadata = {
719
+ level,
720
+ ...(hContent.alignment ? { alignment: hContent.alignment } : {}),
721
+ ...(hContent.style ? { style: hContent.style } : {}),
722
+ ...(hContent.anchorIds?.length ? { anchorIds: hContent.anchorIds } : {})
723
+ };
724
+ const nodeId = node.getAttribute("xml:id") || node.getAttribute("text:id");
725
+ if (nodeId) {
726
+ if (!metadata.anchorIds)
727
+ metadata.anchorIds = [];
728
+ metadata.anchorIds.push(nodeId);
729
+ }
678
730
  const hNode = {
679
731
  type: 'heading',
680
732
  text: hContent.text,
681
733
  children: hContent.children,
682
- metadata: {
683
- level,
684
- ...(hContent.alignment ? { alignment: hContent.alignment } : {}),
685
- ...(hContent.style ? { style: hContent.style } : {})
686
- }
734
+ metadata
687
735
  };
688
736
  if (config.includeRawContent) {
689
737
  hNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
@@ -694,6 +742,20 @@ const parseOpenOffice = async (buffer, config) => {
694
742
  else if (node.tagName === "table:table") {
695
743
  // Parse table with proper structure
696
744
  const tableNode = parseTable(node, paragraphStyleMap, styleMap, config, sourceXml);
745
+ if (asSheet) {
746
+ tableNode.type = 'sheet';
747
+ const sheetName = node.getAttribute("table:name");
748
+ if (sheetName) {
749
+ tableNode.metadata = { ...tableNode.metadata, sheetName };
750
+ }
751
+ }
752
+ const tableId = node.getAttribute("xml:id") || node.getAttribute("table:name");
753
+ if (tableId) {
754
+ if (!tableNode.metadata)
755
+ tableNode.metadata = {};
756
+ tableNode.metadata.anchorIds = tableNode.metadata.anchorIds || [];
757
+ tableNode.metadata.anchorIds.push(tableId);
758
+ }
697
759
  if (config.includeRawContent) {
698
760
  tableNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
699
761
  }
@@ -720,9 +782,8 @@ const parseOpenOffice = async (buffer, config) => {
720
782
  parentNode = parentNode.parentNode;
721
783
  }
722
784
  }
723
- // Try to find list style in automatic styles to determine type and visibility
785
+ // Try to find list style in automatic styles or styles.xml to determine type and visibility
724
786
  if (styleNameToCheck) {
725
- const automaticStyles = (0, xmlUtils_js_1.getFirstElementByTagName)((0, xmlUtils_js_1.parseXmlString)(mainContentFile?.content.toString() || ''), "office:automatic-styles");
726
787
  if (automaticStyles) {
727
788
  const listStyles = (0, xmlUtils_js_1.getElementsByTagName)(automaticStyles, "text:list-style");
728
789
  for (const listStyle of listStyles) {
@@ -739,32 +800,32 @@ const parseOpenOffice = async (buffer, config) => {
739
800
  listType = 'unordered';
740
801
  isVisible = bulletLevels.some(l => !!l.getAttribute("text:bullet-char"));
741
802
  }
742
- if (imageLevels.length > 0)
803
+ else if (imageLevels.length > 0) {
804
+ listType = 'unordered';
743
805
  isVisible = true;
806
+ }
744
807
  break;
745
808
  }
746
809
  }
747
810
  }
748
- // Also check in styles.xml if still unordered and hidden
749
- if (stylesFile && !isVisible) {
750
- const stylesXml = (0, xmlUtils_js_1.parseXmlString)(stylesFile.content.toString());
751
- const listStyles = (0, xmlUtils_js_1.getElementsByTagName)(stylesXml, "text:list-style");
752
- for (const listStyle of listStyles) {
753
- if (listStyle.getAttribute("style:name") === styleNameToCheck) {
754
- const bulletLevels = (0, xmlUtils_js_1.getElementsByTagName)(listStyle, "text:list-level-style-bullet");
755
- const numberLevels = (0, xmlUtils_js_1.getElementsByTagName)(listStyle, "text:list-level-style-number");
756
- const imageLevels = (0, xmlUtils_js_1.getElementsByTagName)(listStyle, "text:list-level-style-image");
757
- if (numberLevels.length > 0) {
758
- listType = 'ordered';
759
- isVisible = numberLevels.some(l => !!l.getAttribute("style:num-format"));
760
- }
761
- else if (bulletLevels.length > 0) {
762
- listType = 'unordered';
763
- isVisible = bulletLevels.some(l => !!l.getAttribute("text:bullet-char"));
811
+ if (!isVisible && stylesDom) {
812
+ const officeStyles = (0, xmlUtils_js_1.getFirstElementByTagName)(stylesDom, "office:styles");
813
+ if (officeStyles) {
814
+ const listStyles = (0, xmlUtils_js_1.getElementsByTagName)(officeStyles, "text:list-style");
815
+ for (const listStyle of listStyles) {
816
+ if (listStyle.getAttribute("style:name") === styleNameToCheck) {
817
+ const bulletLevels = (0, xmlUtils_js_1.getElementsByTagName)(listStyle, "text:list-level-style-bullet");
818
+ const numberLevels = (0, xmlUtils_js_1.getElementsByTagName)(listStyle, "text:list-level-style-number");
819
+ if (numberLevels.length > 0) {
820
+ listType = 'ordered';
821
+ isVisible = numberLevels.some(l => !!l.getAttribute("style:num-format"));
822
+ }
823
+ else if (bulletLevels.length > 0) {
824
+ listType = 'unordered';
825
+ isVisible = bulletLevels.some(l => !!l.getAttribute("text:bullet-char"));
826
+ }
827
+ break;
764
828
  }
765
- if (imageLevels.length > 0)
766
- isVisible = true;
767
- break;
768
829
  }
769
830
  }
770
831
  }
@@ -919,14 +980,19 @@ const parseOpenOffice = async (buffer, config) => {
919
980
  const parts = imageHref.split('/');
920
981
  imageHref = parts[parts.length - 1];
921
982
  }
983
+ const metadata = {
984
+ attachmentName: imageHref,
985
+ ...(altText ? { altText } : {})
986
+ };
987
+ const frameId = node.getAttribute("xml:id") || node.getAttribute("draw:name");
988
+ if (frameId) {
989
+ metadata.anchorIds = [frameId];
990
+ }
922
991
  const imageNode = {
923
992
  type: 'image',
924
993
  text: '',
925
994
  children: [],
926
- metadata: {
927
- attachmentName: imageHref,
928
- ...(altText ? { altText } : {})
929
- }
995
+ metadata
930
996
  };
931
997
  if (config.includeRawContent) {
932
998
  imageNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
@@ -1259,10 +1325,10 @@ const parseOpenOffice = async (buffer, config) => {
1259
1325
  if (config.ocr) {
1260
1326
  if (attachment.mimeType.startsWith('image/')) {
1261
1327
  try {
1262
- attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(media.content, { language: config.ocrLanguage, ...config.ocrConfig })).trim();
1328
+ attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(media.content, { ...config.ocrConfig })).trim();
1263
1329
  }
1264
1330
  catch (e) {
1265
- (0, errorUtils_js_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
1331
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.OCR_FAILED, config, attachment.name, e);
1266
1332
  }
1267
1333
  }
1268
1334
  }
@@ -1399,7 +1465,7 @@ const parseOpenOffice = async (buffer, config) => {
1399
1465
  node.text = attachment.ocrText;
1400
1466
  }
1401
1467
  if (attachment.chartData && node.type === 'chart') {
1402
- node.text = attachment.chartData.rawTexts.join(config.newlineDelimiter ?? '\n');
1468
+ node.text = attachment.chartData.rawTexts.join(config.newlineDelimiter);
1403
1469
  }
1404
1470
  }
1405
1471
  }
@@ -1433,32 +1499,27 @@ const parseOpenOffice = async (buffer, config) => {
1433
1499
  if (config.putNotesAtLast && notes.length > 0) {
1434
1500
  content.push(...notes);
1435
1501
  }
1436
- return {
1437
- type: fileType,
1438
- metadata: {
1439
- ...metadata,
1440
- styleMap: combinedStyleMap
1441
- },
1442
- content: content,
1443
- attachments: attachments,
1444
- toText: () => content.map(c => {
1445
- const getText = (node) => {
1446
- let t = '';
1447
- if (node.children && node.children.length > 0) {
1448
- // Check if children have their own children (container vs leaf)
1449
- // If children are leaf nodes (text/image), join with empty string
1450
- // If children are container nodes (paragraphs/rows), join with newline
1451
- const hasGrandChildren = node.children.some(child => child.children && child.children.length > 0);
1452
- const separator = hasGrandChildren ? (config.newlineDelimiter ?? '\n') : '';
1453
- t += node.children.map(getText).filter(t => t != '').join(separator);
1454
- }
1455
- else {
1456
- t += node.text || '';
1457
- }
1458
- return t;
1459
- };
1460
- return getText(c);
1461
- }).filter(t => t != '').join(config.newlineDelimiter ?? '\n')
1462
- };
1502
+ const toTextSync = () => content.map(c => {
1503
+ const getText = (node) => {
1504
+ let t = '';
1505
+ if (node.children && node.children.length > 0) {
1506
+ // Check if children have their own children (container vs leaf)
1507
+ // If children are leaf nodes (text/image), join with empty string
1508
+ // If children are container nodes (paragraphs/rows), join with newline
1509
+ const hasGrandChildren = node.children.some(child => child.children && child.children.length > 0);
1510
+ const separator = hasGrandChildren ? config.newlineDelimiter : '';
1511
+ t += node.children.map(getText).filter(t => t != '').join(separator);
1512
+ }
1513
+ else {
1514
+ t += node.text || '';
1515
+ }
1516
+ return t;
1517
+ };
1518
+ return getText(c);
1519
+ }).filter(t => t != '').join(config.newlineDelimiter);
1520
+ return (0, astUtils_js_1.createAST)(fileType, {
1521
+ ...metadata,
1522
+ styleMap: combinedStyleMap
1523
+ }, content, attachments, config, toTextSync);
1463
1524
  };
1464
1525
  exports.parseOpenOffice = parseOpenOffice;
@@ -56,7 +56,7 @@
56
56
  * @see https://mozilla.github.io/pdf.js/ PDF.js documentation
57
57
  * @see https://www.adobe.com/devnet/pdf/pdf_reference.html PDF Reference
58
58
  */
59
- import { OfficeParserAST, OfficeParserConfig } from '../types.js';
59
+ import { FullOfficeParserConfig, OfficeParserAST } from '../types.js';
60
60
  /**
61
61
  * Parses a PDF file and extracts content.
62
62
  *
@@ -64,4 +64,4 @@ import { OfficeParserAST, OfficeParserConfig } from '../types.js';
64
64
  * @param config - Parser configuration
65
65
  * @returns Promise resolving to the parsed AST
66
66
  */
67
- export declare const parsePdf: (buffer: Buffer, config: OfficeParserConfig) => Promise<OfficeParserAST>;
67
+ export declare const parsePdf: (buffer: Buffer, config: FullOfficeParserConfig) => Promise<OfficeParserAST>;
@@ -59,12 +59,15 @@
59
59
  */
60
60
  Object.defineProperty(exports, "__esModule", { value: true });
61
61
  exports.parsePdf = void 0;
62
+ const defaults_js_1 = require("../defaults.js");
63
+ const types_js_1 = require("../types.js");
64
+ const astUtils_js_1 = require("../utils/astUtils.js");
65
+ const dateUtils_js_1 = require("../utils/dateUtils.js");
66
+ const envUtils_js_1 = require("../utils/envUtils.js");
62
67
  const errorUtils_js_1 = require("../utils/errorUtils.js");
63
68
  const imageUtils_js_1 = require("../utils/imageUtils.js");
64
- const ocrUtils_js_1 = require("../utils/ocrUtils.js");
65
69
  const moduleLoader_js_1 = require("../utils/moduleLoader.js");
66
- const dateUtils_js_1 = require("../utils/dateUtils.js");
67
- const envUtils_js_1 = require("../utils/envUtils.js");
70
+ const ocrUtils_js_1 = require("../utils/ocrUtils.js");
68
71
  /** Type guard for TextItem in PDF.js 5.x */
69
72
  function isTextItem(item) {
70
73
  return item && typeof item.str === 'string' && Array.isArray(item.transform) && item.transform.length >= 6;
@@ -265,29 +268,38 @@ function convertToRgbaBuffer(data, width, height, kind) {
265
268
  const parsePdf = async (buffer, config) => {
266
269
  const pdfjs = await (0, moduleLoader_js_1.loadPdfJs)();
267
270
  // Configure worker
268
- if (config.pdfWorkerSrc) {
269
- pdfjs.GlobalWorkerOptions.workerSrc = config.pdfWorkerSrc;
271
+ const workerSrc = config.pdfWorkerSrc;
272
+ if (envUtils_js_1.isBrowser) {
273
+ pdfjs.GlobalWorkerOptions.workerSrc = workerSrc;
270
274
  }
271
275
  else {
272
- // Fallbacks when no workerSrc is provided
273
- if (envUtils_js_1.isBrowser) {
274
- // Browser: Default to CDN
275
- pdfjs.GlobalWorkerOptions.workerSrc = `https://unpkg.com/pdfjs-dist@${pdfjs.version}/build/pdf.worker.min.mjs`;
276
+ // Node.js: Try to auto-resolve local worker path to avoid remote fetch errors
277
+ (0, envUtils_js_1.assertNode)('pdf-worker-auto-resolution');
278
+ let resolved = false;
279
+ // If the user provided a custom path (not the default CDN one), use it.
280
+ // Otherwise, try to find it locally.
281
+ if (workerSrc !== defaults_js_1.DEFAULT_OFFICE_PARSER_CONFIG.pdfWorkerSrc && workerSrc !== '') {
282
+ pdfjs.GlobalWorkerOptions.workerSrc = workerSrc;
283
+ resolved = true;
276
284
  }
277
285
  else {
278
- // Node.js: Try to auto-resolve local worker path to avoid remote fetch errors
279
- (0, envUtils_js_1.assertNode)('pdf-worker-auto-resolution');
280
286
  try {
281
287
  // We use require.resolve to find the exact path of the installed package.
282
288
  // @ts-ignore - 'require' is available in Node.js/CommonJS environment
283
- const workerPath = require.resolve('pdfjs-dist/legacy/build/pdf.worker.mjs');
284
- pdfjs.GlobalWorkerOptions.workerSrc = workerPath;
289
+ const localWorkerPath = require.resolve('pdfjs-dist/legacy/build/pdf.worker.mjs');
290
+ // Use file:// URL for the worker source in Node.js to ensure compatibility with ESM-native PDF.js 5.x
291
+ // We use dynamic import for 'url' to avoid breaking browser bundles
292
+ const { pathToFileURL } = await import('url');
293
+ pdfjs.GlobalWorkerOptions.workerSrc = pathToFileURL(localWorkerPath).href;
294
+ resolved = true;
285
295
  }
286
296
  catch (e) {
287
- if (config.outputErrorToConsole)
288
- console.warn("[PdfParser] Could not auto-resolve local worker path:", e);
297
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.PDF_WORKER_FALLBACK, config, undefined, e);
289
298
  }
290
299
  }
300
+ if (!resolved) {
301
+ pdfjs.GlobalWorkerOptions.workerSrc = workerSrc;
302
+ }
291
303
  }
292
304
  const uint8Array = new Uint8Array(buffer);
293
305
  const loadingTask = pdfjs.getDocument({
@@ -302,7 +314,7 @@ const parsePdf = async (buffer, config) => {
302
314
  catch (e) {
303
315
  const message = e instanceof Error ? e.message : String(e);
304
316
  if (message.includes('workerSrc') || message.includes('No "GlobalWorkerOptions.workerSrc" specified')) {
305
- throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.PDF_WORKER_MISSING, config);
317
+ throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.PDF_WORKER_MISSING, config);
306
318
  }
307
319
  throw e;
308
320
  }
@@ -382,8 +394,7 @@ const parsePdf = async (buffer, config) => {
382
394
  }
383
395
  }
384
396
  catch (e) {
385
- if (config.outputErrorToConsole)
386
- console.error("Error extracting embedded attachments:", e);
397
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.ATTACHMENT_EXTRACTION_FAILED, config, undefined, e);
387
398
  }
388
399
  // --- First Pass: Collect all items for font statistics ---
389
400
  for (let i = 1; i <= numPages; i++) {
@@ -396,13 +407,13 @@ const parsePdf = async (buffer, config) => {
396
407
  textContent = await page.getTextContent();
397
408
  }
398
409
  catch (e) {
399
- if (config.outputErrorToConsole)
400
- console.warn(`[PdfParser] Error loading page ${i}:`, e);
410
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.PAGE_LOAD_FAILED, config, i, e);
401
411
  // Push empty items to maintain index alignment for second pass
402
412
  allPageItems.push(pageItems);
403
413
  continue;
404
414
  }
405
415
  const commonObjs = page.commonObjs;
416
+ const fontCache = new Map();
406
417
  for (const item of textContent.items) {
407
418
  // PDF.js 5.x: textContent.items can contain TextMarkedContent which lack
408
419
  // 'str' and 'transform'. Skip these to avoid crashes and page skipping.
@@ -422,11 +433,15 @@ const parsePdf = async (buffer, config) => {
422
433
  if (textItem.fontName && commonObjs) {
423
434
  try {
424
435
  if (commonObjs.has(textItem.fontName)) {
425
- // Use callback-based get to ensure safe resolution
426
- const fontData = await new Promise((resolve) => {
427
- // @ts-ignore - commonObjs.get is callback-based in legacy builds
428
- commonObjs.get(textItem.fontName, (data) => resolve(data));
429
- });
436
+ let fontData = fontCache.get(textItem.fontName);
437
+ if (!fontData) {
438
+ // Use callback-based get to ensure safe resolution
439
+ fontData = await new Promise((resolve) => {
440
+ // @ts-ignore - commonObjs.get is callback-based in legacy builds
441
+ commonObjs.get(textItem.fontName, (data) => resolve(data));
442
+ });
443
+ fontCache.set(textItem.fontName, fontData);
444
+ }
430
445
  if (fontData?.name && typeof fontData.name === 'string') {
431
446
  // Remove PDF subset prefix (6 uppercase letters + '+')
432
447
  fontName = fontData.name.replace(/^[A-Z]{6}\+/, '');
@@ -485,9 +500,7 @@ const parsePdf = async (buffer, config) => {
485
500
  });
486
501
  }
487
502
  catch (e) {
488
- if (config.outputErrorToConsole) {
489
- console.error(`[PdfParser] Failed to load dependency ${dep}:`, e);
490
- }
503
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.DEPENDENCY_LOAD_FAILED, config, dep, e);
491
504
  }
492
505
  }
493
506
  }
@@ -507,7 +520,7 @@ const parsePdf = async (buffer, config) => {
507
520
  targetObjs.get(imgName, (data) => resolve(data));
508
521
  });
509
522
  // Browser-specific: Handle ImageBitmap if data is missing
510
- if (typeof window !== 'undefined' && !imgObj.data && imgObj.bitmap) {
523
+ if (envUtils_js_1.isBrowser && !imgObj.data && imgObj.bitmap) {
511
524
  try {
512
525
  const canvas = document.createElement('canvas');
513
526
  canvas.width = imgObj.width;
@@ -520,8 +533,7 @@ const parsePdf = async (buffer, config) => {
520
533
  }
521
534
  }
522
535
  catch (e) {
523
- if (config.outputErrorToConsole)
524
- console.error(`[PdfParser] Failed to extract from ImageBitmap:`, e);
536
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.IMAGE_PROCESSING_FAILED, config, undefined, e);
525
537
  }
526
538
  }
527
539
  if (imgObj?.data && imgObj.width > 0 && imgObj.height > 0) {
@@ -554,8 +566,7 @@ const parsePdf = async (buffer, config) => {
554
566
  }
555
567
  }
556
568
  catch (e) {
557
- if (config.outputErrorToConsole)
558
- console.error(`Error extracting images from page ${i}:`, e);
569
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.IMAGE_EXTRACTION_FAILED, config, `from page ${i}`, e);
559
570
  }
560
571
  }
561
572
  allPageItems.push(pageItems);
@@ -570,8 +581,7 @@ const parsePdf = async (buffer, config) => {
570
581
  page = await pdfDocument.getPage(pageNum);
571
582
  }
572
583
  catch (e) {
573
- if (config.outputErrorToConsole)
574
- console.warn(`[PdfParser] Error loading page ${pageNum} in second pass:`, e);
584
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.PAGE_LOAD_FAILED, config, pageNum, e);
575
585
  continue;
576
586
  }
577
587
  const pageItems = allPageItems[i];
@@ -594,8 +604,7 @@ const parsePdf = async (buffer, config) => {
594
604
  }
595
605
  }
596
606
  catch (e) {
597
- if (config.outputErrorToConsole)
598
- console.error(`Error extracting annotations from page ${pageNum}:`, e);
607
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.ANNOTATION_EXTRACTION_FAILED, config, pageNum, e);
599
608
  }
600
609
  // Sort items: Y descending (top to bottom), then X ascending (left to right)
601
610
  pageItems.sort((a, b) => {
@@ -744,12 +753,11 @@ const parsePdf = async (buffer, config) => {
744
753
  try {
745
754
  // Skip OCR for very small images/artifacts (e.g. < 10px) to avoid Tesseract warnings
746
755
  if (item.width >= 10 && item.height >= 10) {
747
- attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(bmpBuffer, { language: config.ocrLanguage, ...config.ocrConfig })).trim();
756
+ attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(bmpBuffer, { ...config.ocrConfig })).trim();
748
757
  }
749
758
  }
750
759
  catch (e) {
751
- if (config.outputErrorToConsole)
752
- console.error(`OCR failed for ${attachmentName}:`, e);
760
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.OCR_FAILED, config, attachmentName, e);
753
761
  }
754
762
  }
755
763
  attachments.push(attachment);
@@ -764,7 +772,7 @@ const parsePdf = async (buffer, config) => {
764
772
  });
765
773
  }
766
774
  catch (e) {
767
- (0, errorUtils_js_1.logWarning)(`Failed to process image ${attachmentName}:`, config, e);
775
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.IMAGE_EXTRACTION_FAILED, config, attachmentName, e);
768
776
  }
769
777
  }
770
778
  }
@@ -777,17 +785,12 @@ const parsePdf = async (buffer, config) => {
777
785
  content.push({
778
786
  type: 'page',
779
787
  children: pageContent,
780
- text: pageContent.map(node => node.text).join(config.newlineDelimiter ?? '\n\n'),
788
+ text: pageContent.map(node => node.text).join(config.newlineDelimiter),
781
789
  metadata: { pageNumber: pageNum }
782
790
  });
783
791
  }
784
- return {
785
- type: 'pdf',
786
- metadata: metadata,
787
- content: content,
788
- attachments: attachments,
789
- toText: () => content.map(c => c.text).join(config.newlineDelimiter ?? '\n\n')
790
- };
792
+ const toTextSync = () => content.map(c => c.text).join(config.newlineDelimiter);
793
+ return (0, astUtils_js_1.createAST)('pdf', metadata, content, attachments, config, toTextSync);
791
794
  };
792
795
  exports.parsePdf = parsePdf;
793
796
  /**
@@ -21,7 +21,7 @@
21
21
  * @module PowerPointParser
22
22
  * @see https://www.ecma-international.org/publications-and-standards/standards/ecma-376/
23
23
  */
24
- import { OfficeParserAST, OfficeParserConfig } from '../types.js';
24
+ import { FullOfficeParserConfig, OfficeParserAST } from '../types.js';
25
25
  /**
26
26
  * Parses a PowerPoint presentation (.pptx) and extracts slides and notes.
27
27
  *
@@ -29,4 +29,4 @@ import { OfficeParserAST, OfficeParserConfig } from '../types.js';
29
29
  * @param config - Parser configuration
30
30
  * @returns A promise resolving to the parsed AST
31
31
  */
32
- export declare const parsePowerPoint: (buffer: Buffer, config: OfficeParserConfig) => Promise<OfficeParserAST>;
32
+ export declare const parsePowerPoint: (buffer: Buffer, config: FullOfficeParserConfig) => Promise<OfficeParserAST>;