officeparser 6.1.0 → 7.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (70) hide show
  1. package/README.md +284 -86
  2. package/dist/OfficeConverter.d.ts +46 -0
  3. package/dist/OfficeConverter.js +72 -0
  4. package/dist/OfficeGenerator.d.ts +19 -0
  5. package/dist/OfficeGenerator.js +48 -0
  6. package/dist/OfficeParser.d.ts +6 -0
  7. package/dist/OfficeParser.js +55 -28
  8. package/dist/cli.d.ts +3 -1
  9. package/dist/cli.js +107 -22
  10. package/dist/defaults.d.ts +41 -0
  11. package/dist/defaults.js +172 -0
  12. package/dist/generators/BaseGenerator.d.ts +58 -0
  13. package/dist/generators/BaseGenerator.js +107 -0
  14. package/dist/generators/ChunkingGenerator.d.ts +81 -0
  15. package/dist/generators/ChunkingGenerator.js +683 -0
  16. package/dist/generators/CsvGenerator.d.ts +30 -0
  17. package/dist/generators/CsvGenerator.js +233 -0
  18. package/dist/generators/HtmlGenerator.d.ts +37 -0
  19. package/dist/generators/HtmlGenerator.js +1013 -0
  20. package/dist/generators/MarkdownGenerator.d.ts +59 -0
  21. package/dist/generators/MarkdownGenerator.js +481 -0
  22. package/dist/generators/PdfGenerator.d.ts +22 -0
  23. package/dist/generators/PdfGenerator.js +118 -0
  24. package/dist/generators/RtfGenerator.d.ts +15 -0
  25. package/dist/generators/RtfGenerator.js +208 -0
  26. package/dist/generators/TextGenerator.d.ts +13 -0
  27. package/dist/generators/TextGenerator.js +108 -0
  28. package/dist/index.d.ts +11 -3
  29. package/dist/index.js +17 -2
  30. package/dist/index.mjs +2 -2
  31. package/dist/officeparser.browser.d.ts +878 -5
  32. package/dist/officeparser.browser.iife.js +703 -49
  33. package/dist/officeparser.browser.mjs +703 -49
  34. package/dist/parsers/CsvParser.d.ts +9 -0
  35. package/dist/parsers/CsvParser.js +110 -0
  36. package/dist/parsers/ExcelParser.d.ts +2 -2
  37. package/dist/parsers/ExcelParser.js +145 -114
  38. package/dist/parsers/HtmlParser.d.ts +2 -0
  39. package/dist/parsers/HtmlParser.js +539 -0
  40. package/dist/parsers/MarkdownParser.d.ts +2 -0
  41. package/dist/parsers/MarkdownParser.js +360 -0
  42. package/dist/parsers/OpenOfficeParser.d.ts +2 -2
  43. package/dist/parsers/OpenOfficeParser.js +237 -128
  44. package/dist/parsers/PdfParser.d.ts +2 -2
  45. package/dist/parsers/PdfParser.js +52 -49
  46. package/dist/parsers/PowerPointParser.d.ts +2 -2
  47. package/dist/parsers/PowerPointParser.js +132 -123
  48. package/dist/parsers/RtfParser.d.ts +22 -2
  49. package/dist/parsers/RtfParser.js +1398 -1282
  50. package/dist/parsers/WordParser.d.ts +3 -2
  51. package/dist/parsers/WordParser.js +333 -115
  52. package/dist/sbom.cdx.json +103 -103
  53. package/dist/types.d.ts +833 -5
  54. package/dist/types.js +71 -0
  55. package/dist/utils/astUtils.d.ts +16 -0
  56. package/dist/utils/astUtils.js +32 -0
  57. package/dist/utils/configUtils.d.ts +26 -0
  58. package/dist/utils/configUtils.js +140 -0
  59. package/dist/utils/envUtils.js +56 -2
  60. package/dist/utils/errorUtils.d.ts +17 -29
  61. package/dist/utils/errorUtils.js +109 -52
  62. package/dist/utils/moduleLoader.js +15 -9
  63. package/dist/utils/ocrUtils.js +2 -1
  64. package/dist/utils/sheetUtils.d.ts +7 -0
  65. package/dist/utils/sheetUtils.js +35 -0
  66. package/dist/utils/styleMapper.d.ts +36 -0
  67. package/dist/utils/styleMapper.js +224 -0
  68. package/dist/utils/xmlUtils.d.ts +0 -8
  69. package/dist/utils/xmlUtils.js +2 -1
  70. package/package.json +28 -9
@@ -59,12 +59,15 @@
59
59
  */
60
60
  Object.defineProperty(exports, "__esModule", { value: true });
61
61
  exports.parsePdf = void 0;
62
+ const defaults_js_1 = require("../defaults.js");
63
+ const types_js_1 = require("../types.js");
64
+ const astUtils_js_1 = require("../utils/astUtils.js");
65
+ const dateUtils_js_1 = require("../utils/dateUtils.js");
66
+ const envUtils_js_1 = require("../utils/envUtils.js");
62
67
  const errorUtils_js_1 = require("../utils/errorUtils.js");
63
68
  const imageUtils_js_1 = require("../utils/imageUtils.js");
64
- const ocrUtils_js_1 = require("../utils/ocrUtils.js");
65
69
  const moduleLoader_js_1 = require("../utils/moduleLoader.js");
66
- const dateUtils_js_1 = require("../utils/dateUtils.js");
67
- const envUtils_js_1 = require("../utils/envUtils.js");
70
+ const ocrUtils_js_1 = require("../utils/ocrUtils.js");
68
71
  /** Type guard for TextItem in PDF.js 5.x */
69
72
  function isTextItem(item) {
70
73
  return item && typeof item.str === 'string' && Array.isArray(item.transform) && item.transform.length >= 6;
@@ -265,29 +268,38 @@ function convertToRgbaBuffer(data, width, height, kind) {
265
268
  const parsePdf = async (buffer, config) => {
266
269
  const pdfjs = await (0, moduleLoader_js_1.loadPdfJs)();
267
270
  // Configure worker
268
- if (config.pdfWorkerSrc) {
269
- pdfjs.GlobalWorkerOptions.workerSrc = config.pdfWorkerSrc;
271
+ const workerSrc = config.pdfWorkerSrc;
272
+ if (envUtils_js_1.isBrowser) {
273
+ pdfjs.GlobalWorkerOptions.workerSrc = workerSrc;
270
274
  }
271
275
  else {
272
- // Fallbacks when no workerSrc is provided
273
- if (envUtils_js_1.isBrowser) {
274
- // Browser: Default to CDN
275
- pdfjs.GlobalWorkerOptions.workerSrc = `https://unpkg.com/pdfjs-dist@${pdfjs.version}/build/pdf.worker.min.mjs`;
276
+ // Node.js: Try to auto-resolve local worker path to avoid remote fetch errors
277
+ (0, envUtils_js_1.assertNode)('pdf-worker-auto-resolution');
278
+ let resolved = false;
279
+ // If the user provided a custom path (not the default CDN one), use it.
280
+ // Otherwise, try to find it locally.
281
+ if (workerSrc !== defaults_js_1.DEFAULT_OFFICE_PARSER_CONFIG.pdfWorkerSrc && workerSrc !== '') {
282
+ pdfjs.GlobalWorkerOptions.workerSrc = workerSrc;
283
+ resolved = true;
276
284
  }
277
285
  else {
278
- // Node.js: Try to auto-resolve local worker path to avoid remote fetch errors
279
- (0, envUtils_js_1.assertNode)('pdf-worker-auto-resolution');
280
286
  try {
281
287
  // We use require.resolve to find the exact path of the installed package.
282
288
  // @ts-ignore - 'require' is available in Node.js/CommonJS environment
283
- const workerPath = require.resolve('pdfjs-dist/legacy/build/pdf.worker.mjs');
284
- pdfjs.GlobalWorkerOptions.workerSrc = workerPath;
289
+ const localWorkerPath = require.resolve('pdfjs-dist/legacy/build/pdf.worker.mjs');
290
+ // Use file:// URL for the worker source in Node.js to ensure compatibility with ESM-native PDF.js 5.x
291
+ // We use dynamic import for 'url' to avoid breaking browser bundles
292
+ const { pathToFileURL } = await import('url');
293
+ pdfjs.GlobalWorkerOptions.workerSrc = pathToFileURL(localWorkerPath).href;
294
+ resolved = true;
285
295
  }
286
296
  catch (e) {
287
- if (config.outputErrorToConsole)
288
- console.warn("[PdfParser] Could not auto-resolve local worker path:", e);
297
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.PDF_WORKER_FALLBACK, config, undefined, e);
289
298
  }
290
299
  }
300
+ if (!resolved) {
301
+ pdfjs.GlobalWorkerOptions.workerSrc = workerSrc;
302
+ }
291
303
  }
292
304
  const uint8Array = new Uint8Array(buffer);
293
305
  const loadingTask = pdfjs.getDocument({
@@ -302,7 +314,7 @@ const parsePdf = async (buffer, config) => {
302
314
  catch (e) {
303
315
  const message = e instanceof Error ? e.message : String(e);
304
316
  if (message.includes('workerSrc') || message.includes('No "GlobalWorkerOptions.workerSrc" specified')) {
305
- throw (0, errorUtils_js_1.getOfficeError)(errorUtils_js_1.OfficeErrorType.PDF_WORKER_MISSING, config);
317
+ throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.PDF_WORKER_MISSING, config);
306
318
  }
307
319
  throw e;
308
320
  }
@@ -382,8 +394,7 @@ const parsePdf = async (buffer, config) => {
382
394
  }
383
395
  }
384
396
  catch (e) {
385
- if (config.outputErrorToConsole)
386
- console.error("Error extracting embedded attachments:", e);
397
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.ATTACHMENT_EXTRACTION_FAILED, config, undefined, e);
387
398
  }
388
399
  // --- First Pass: Collect all items for font statistics ---
389
400
  for (let i = 1; i <= numPages; i++) {
@@ -396,13 +407,13 @@ const parsePdf = async (buffer, config) => {
396
407
  textContent = await page.getTextContent();
397
408
  }
398
409
  catch (e) {
399
- if (config.outputErrorToConsole)
400
- console.warn(`[PdfParser] Error loading page ${i}:`, e);
410
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.PAGE_LOAD_FAILED, config, i, e);
401
411
  // Push empty items to maintain index alignment for second pass
402
412
  allPageItems.push(pageItems);
403
413
  continue;
404
414
  }
405
415
  const commonObjs = page.commonObjs;
416
+ const fontCache = new Map();
406
417
  for (const item of textContent.items) {
407
418
  // PDF.js 5.x: textContent.items can contain TextMarkedContent which lack
408
419
  // 'str' and 'transform'. Skip these to avoid crashes and page skipping.
@@ -422,11 +433,15 @@ const parsePdf = async (buffer, config) => {
422
433
  if (textItem.fontName && commonObjs) {
423
434
  try {
424
435
  if (commonObjs.has(textItem.fontName)) {
425
- // Use callback-based get to ensure safe resolution
426
- const fontData = await new Promise((resolve) => {
427
- // @ts-ignore - commonObjs.get is callback-based in legacy builds
428
- commonObjs.get(textItem.fontName, (data) => resolve(data));
429
- });
436
+ let fontData = fontCache.get(textItem.fontName);
437
+ if (!fontData) {
438
+ // Use callback-based get to ensure safe resolution
439
+ fontData = await new Promise((resolve) => {
440
+ // @ts-ignore - commonObjs.get is callback-based in legacy builds
441
+ commonObjs.get(textItem.fontName, (data) => resolve(data));
442
+ });
443
+ fontCache.set(textItem.fontName, fontData);
444
+ }
430
445
  if (fontData?.name && typeof fontData.name === 'string') {
431
446
  // Remove PDF subset prefix (6 uppercase letters + '+')
432
447
  fontName = fontData.name.replace(/^[A-Z]{6}\+/, '');
@@ -485,9 +500,7 @@ const parsePdf = async (buffer, config) => {
485
500
  });
486
501
  }
487
502
  catch (e) {
488
- if (config.outputErrorToConsole) {
489
- console.error(`[PdfParser] Failed to load dependency ${dep}:`, e);
490
- }
503
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.DEPENDENCY_LOAD_FAILED, config, dep, e);
491
504
  }
492
505
  }
493
506
  }
@@ -507,7 +520,7 @@ const parsePdf = async (buffer, config) => {
507
520
  targetObjs.get(imgName, (data) => resolve(data));
508
521
  });
509
522
  // Browser-specific: Handle ImageBitmap if data is missing
510
- if (typeof window !== 'undefined' && !imgObj.data && imgObj.bitmap) {
523
+ if (envUtils_js_1.isBrowser && !imgObj.data && imgObj.bitmap) {
511
524
  try {
512
525
  const canvas = document.createElement('canvas');
513
526
  canvas.width = imgObj.width;
@@ -520,8 +533,7 @@ const parsePdf = async (buffer, config) => {
520
533
  }
521
534
  }
522
535
  catch (e) {
523
- if (config.outputErrorToConsole)
524
- console.error(`[PdfParser] Failed to extract from ImageBitmap:`, e);
536
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.IMAGE_PROCESSING_FAILED, config, undefined, e);
525
537
  }
526
538
  }
527
539
  if (imgObj?.data && imgObj.width > 0 && imgObj.height > 0) {
@@ -554,8 +566,7 @@ const parsePdf = async (buffer, config) => {
554
566
  }
555
567
  }
556
568
  catch (e) {
557
- if (config.outputErrorToConsole)
558
- console.error(`Error extracting images from page ${i}:`, e);
569
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.IMAGE_EXTRACTION_FAILED, config, `from page ${i}`, e);
559
570
  }
560
571
  }
561
572
  allPageItems.push(pageItems);
@@ -570,8 +581,7 @@ const parsePdf = async (buffer, config) => {
570
581
  page = await pdfDocument.getPage(pageNum);
571
582
  }
572
583
  catch (e) {
573
- if (config.outputErrorToConsole)
574
- console.warn(`[PdfParser] Error loading page ${pageNum} in second pass:`, e);
584
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.PAGE_LOAD_FAILED, config, pageNum, e);
575
585
  continue;
576
586
  }
577
587
  const pageItems = allPageItems[i];
@@ -594,8 +604,7 @@ const parsePdf = async (buffer, config) => {
594
604
  }
595
605
  }
596
606
  catch (e) {
597
- if (config.outputErrorToConsole)
598
- console.error(`Error extracting annotations from page ${pageNum}:`, e);
607
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.ANNOTATION_EXTRACTION_FAILED, config, pageNum, e);
599
608
  }
600
609
  // Sort items: Y descending (top to bottom), then X ascending (left to right)
601
610
  pageItems.sort((a, b) => {
@@ -744,12 +753,11 @@ const parsePdf = async (buffer, config) => {
744
753
  try {
745
754
  // Skip OCR for very small images/artifacts (e.g. < 10px) to avoid Tesseract warnings
746
755
  if (item.width >= 10 && item.height >= 10) {
747
- attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(bmpBuffer, { language: config.ocrLanguage, ...config.ocrConfig })).trim();
756
+ attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(bmpBuffer, { ...config.ocrConfig })).trim();
748
757
  }
749
758
  }
750
759
  catch (e) {
751
- if (config.outputErrorToConsole)
752
- console.error(`OCR failed for ${attachmentName}:`, e);
760
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.OCR_FAILED, config, attachmentName, e);
753
761
  }
754
762
  }
755
763
  attachments.push(attachment);
@@ -764,7 +772,7 @@ const parsePdf = async (buffer, config) => {
764
772
  });
765
773
  }
766
774
  catch (e) {
767
- (0, errorUtils_js_1.logWarning)(`Failed to process image ${attachmentName}:`, config, e);
775
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.IMAGE_EXTRACTION_FAILED, config, attachmentName, e);
768
776
  }
769
777
  }
770
778
  }
@@ -777,17 +785,12 @@ const parsePdf = async (buffer, config) => {
777
785
  content.push({
778
786
  type: 'page',
779
787
  children: pageContent,
780
- text: pageContent.map(node => node.text).join(config.newlineDelimiter ?? '\n\n'),
788
+ text: pageContent.map(node => node.text).join(config.newlineDelimiter),
781
789
  metadata: { pageNumber: pageNum }
782
790
  });
783
791
  }
784
- return {
785
- type: 'pdf',
786
- metadata: metadata,
787
- content: content,
788
- attachments: attachments,
789
- toText: () => content.map(c => c.text).join(config.newlineDelimiter ?? '\n\n')
790
- };
792
+ const toTextSync = () => content.map(c => c.text).join(config.newlineDelimiter);
793
+ return (0, astUtils_js_1.createAST)('pdf', metadata, content, attachments, config, toTextSync);
791
794
  };
792
795
  exports.parsePdf = parsePdf;
793
796
  /**
@@ -21,7 +21,7 @@
21
21
  * @module PowerPointParser
22
22
  * @see https://www.ecma-international.org/publications-and-standards/standards/ecma-376/
23
23
  */
24
- import { OfficeParserAST, OfficeParserConfig } from '../types.js';
24
+ import { FullOfficeParserConfig, OfficeParserAST } from '../types.js';
25
25
  /**
26
26
  * Parses a PowerPoint presentation (.pptx) and extracts slides and notes.
27
27
  *
@@ -29,4 +29,4 @@ import { OfficeParserAST, OfficeParserConfig } from '../types.js';
29
29
  * @param config - Parser configuration
30
30
  * @returns A promise resolving to the parsed AST
31
31
  */
32
- export declare const parsePowerPoint: (buffer: Buffer, config: OfficeParserConfig) => Promise<OfficeParserAST>;
32
+ export declare const parsePowerPoint: (buffer: Buffer, config: FullOfficeParserConfig) => Promise<OfficeParserAST>;
@@ -24,6 +24,8 @@
24
24
  */
25
25
  Object.defineProperty(exports, "__esModule", { value: true });
26
26
  exports.parsePowerPoint = void 0;
27
+ const types_js_1 = require("../types.js");
28
+ const astUtils_js_1 = require("../utils/astUtils.js");
27
29
  const chartUtils_js_1 = require("../utils/chartUtils.js");
28
30
  const errorUtils_js_1 = require("../utils/errorUtils.js");
29
31
  const imageUtils_js_1 = require("../utils/imageUtils.js");
@@ -392,117 +394,129 @@ const parsePowerPoint = async (buffer, config) => {
392
394
  if (config.includeRawContent) {
393
395
  pNode.rawContent = (0, xmlUtils_js_1.getRawContent)(p, xmlContentString, config);
394
396
  }
395
- const runs = (0, xmlUtils_js_1.getElementsByTagName)(p, "a:r");
396
- for (let j = 0; j < runs.length; j++) {
397
- const r = runs[j];
398
- const t = (0, xmlUtils_js_1.getFirstElementByTagName)(r, "a:t");
399
- if (t && t.childNodes[0]) {
400
- const textContent = t.childNodes[0].nodeValue || '';
401
- pNode.text += textContent;
402
- const rPr = (0, xmlUtils_js_1.getFirstElementByTagName)(r, "a:rPr");
403
- const formatting = {};
404
- if (rPr) {
405
- if (rPr.getAttribute("b") === "1")
406
- formatting.bold = true;
407
- if (rPr.getAttribute("i") === "1")
408
- formatting.italic = true;
409
- if (rPr.getAttribute("u") === "sng")
410
- formatting.underline = true;
411
- if (rPr.getAttribute("strike") === "sngStrike")
412
- formatting.strikethrough = true;
413
- const sz = rPr.getAttribute("sz");
414
- if (sz)
415
- formatting.size = (parseInt(sz) / 100).toString() + 'pt';
416
- // Color extraction
417
- const solidFill = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:solidFill");
418
- if (solidFill) {
419
- const srgbClr = (0, xmlUtils_js_1.getFirstElementByTagName)(solidFill, "a:srgbClr");
420
- if (srgbClr) {
421
- const val = srgbClr.getAttribute("val");
422
- if (val)
423
- formatting.color = '#' + val;
397
+ // Process all children of <a:p> in order (runs, breaks, fields)
398
+ const children = Array.from(p.childNodes);
399
+ let activeNode = pNode;
400
+ nodes.push(activeNode);
401
+ for (const childNode of children) {
402
+ if (!(0, xmlUtils_js_1.isElement)(childNode))
403
+ continue;
404
+ const element = childNode;
405
+ const tag = element.tagName;
406
+ if (tag === "a:r" || tag === "a:fld") {
407
+ const t = (0, xmlUtils_js_1.getFirstElementByTagName)(element, "a:t");
408
+ if (t && t.childNodes[0]) {
409
+ const textContent = t.childNodes[0].nodeValue || "";
410
+ activeNode.text += textContent;
411
+ const rPr = (0, xmlUtils_js_1.getFirstElementByTagName)(element, "a:rPr");
412
+ const formatting = {};
413
+ if (rPr) {
414
+ if (rPr.getAttribute("b") === "1")
415
+ formatting.bold = true;
416
+ if (rPr.getAttribute("i") === "1")
417
+ formatting.italic = true;
418
+ if (rPr.getAttribute("u") === "sng")
419
+ formatting.underline = true;
420
+ if (rPr.getAttribute("strike") === "sngStrike")
421
+ formatting.strikethrough = true;
422
+ const sz = rPr.getAttribute("sz");
423
+ if (sz)
424
+ formatting.size = (parseInt(sz) / 100).toString() + "pt";
425
+ // Color extraction
426
+ const solidFill = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:solidFill");
427
+ if (solidFill) {
428
+ const srgbClr = (0, xmlUtils_js_1.getFirstElementByTagName)(solidFill, "a:srgbClr");
429
+ if (srgbClr) {
430
+ const val = srgbClr.getAttribute("val");
431
+ if (val)
432
+ formatting.color = "#" + val;
433
+ }
424
434
  }
425
- }
426
- // Highlight extraction
427
- const highlight = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:highlight");
428
- if (highlight) {
429
- const srgbClr = (0, xmlUtils_js_1.getFirstElementByTagName)(highlight, "a:srgbClr");
430
- if (srgbClr) {
431
- const val = srgbClr.getAttribute("val");
432
- if (val)
433
- formatting.backgroundColor = '#' + val;
435
+ // Highlight extraction
436
+ const highlight = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:highlight");
437
+ if (highlight) {
438
+ const srgbClr = (0, xmlUtils_js_1.getFirstElementByTagName)(highlight, "a:srgbClr");
439
+ if (srgbClr) {
440
+ const val = srgbClr.getAttribute("val");
441
+ if (val)
442
+ formatting.backgroundColor = "#" + val;
443
+ }
444
+ }
445
+ // Font family
446
+ const latin = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:latin");
447
+ if (latin) {
448
+ const typeface = latin.getAttribute("typeface");
449
+ if (typeface)
450
+ formatting.font = typeface;
451
+ }
452
+ // Subscript/Superscript
453
+ const baseline = rPr.getAttribute("baseline");
454
+ if (baseline) {
455
+ const baselineVal = parseInt(baseline);
456
+ if (baselineVal < 0)
457
+ formatting.subscript = true;
458
+ if (baselineVal > 0)
459
+ formatting.superscript = true;
434
460
  }
435
461
  }
436
- // Font family
437
- const latin = (0, xmlUtils_js_1.getFirstElementByTagName)(rPr, "a:latin");
438
- if (latin) {
439
- const typeface = latin.getAttribute("typeface");
440
- if (typeface)
441
- formatting.font = typeface;
442
- }
443
- // Subscript/Superscript
444
- const baseline = rPr.getAttribute("baseline");
445
- if (baseline) {
446
- const baselineVal = parseInt(baseline);
447
- if (baselineVal < 0)
448
- formatting.subscript = true;
449
- if (baselineVal > 0)
450
- formatting.superscript = true;
462
+ const textNode = {
463
+ type: 'text',
464
+ text: textContent,
465
+ formatting: formatting
466
+ };
467
+ // Check for Hyperlinks
468
+ const hlinkClick = (0, xmlUtils_js_1.getFirstElementByTagName)(element, "a:hlinkClick");
469
+ if (hlinkClick) {
470
+ const rId = hlinkClick.getAttribute("r:id");
471
+ const action = hlinkClick.getAttribute("action");
472
+ let link;
473
+ let linkType;
474
+ if (rId && slideRelsMap[slideNumber] && slideRelsMap[slideNumber][rId] && slideRelsMap[slideNumber][rId].type === "hyperlink") {
475
+ link = slideRelsMap[slideNumber][rId].target;
476
+ linkType = "external";
477
+ }
478
+ else if (rId && slideRelsMap[slideNumber] && slideRelsMap[slideNumber][rId] && slideRelsMap[slideNumber][rId].type === "slide") {
479
+ link = slideRelsMap[slideNumber][rId].target;
480
+ linkType = "internal";
481
+ }
482
+ else if (action) {
483
+ link = action;
484
+ linkType = "internal";
485
+ }
486
+ if (link) {
487
+ textNode.metadata = { link, linkType };
488
+ }
451
489
  }
490
+ activeNode.children?.push(textNode);
452
491
  }
453
- const textNode = {
454
- type: 'text',
455
- text: textContent,
456
- formatting: formatting
457
- };
458
- // Check for Hyperlinks
459
- const hlinkClick = (0, xmlUtils_js_1.getFirstElementByTagName)(r, "a:hlinkClick");
460
- // Check if this run has a hyperlink click action
461
- if (hlinkClick) {
462
- // Relationship ID for the link
463
- const rId = hlinkClick.getAttribute("r:id");
464
- // Optional action attribute, often for internal jumps
465
- const action = hlinkClick.getAttribute("action");
466
- // Result placeholders
467
- let link;
468
- let linkType;
469
- // Case 1: Relationship exists in slideRelsMap and is a real hyperlink (external URL)
470
- if (rId
471
- && slideRelsMap[slideNumber]
472
- && slideRelsMap[slideNumber][rId]
473
- && slideRelsMap[slideNumber][rId].type === "hyperlink") {
474
- // External URL
475
- link = slideRelsMap[slideNumber][rId].target;
476
- linkType = "external";
477
- }
478
- // Case 2: Relationship exists and is an internal slide reference
479
- else if (rId
480
- && slideRelsMap[slideNumber]
481
- && slideRelsMap[slideNumber][rId]
482
- && slideRelsMap[slideNumber][rId].type === "slide") {
483
- // Example target: ppt/slides/slide3.xml
484
- link = slideRelsMap[slideNumber][rId].target;
485
- linkType = "internal";
486
- }
487
- // Case 3: action attribute like ppaction://hlinksldjump
488
- else if (action) {
489
- link = action;
490
- linkType = "internal";
491
- }
492
- // Assign metadata only if a link was actually discovered
493
- if (link) {
494
- textNode.metadata = { link, linkType };
492
+ }
493
+ else if (tag === "a:br") {
494
+ if (isList) {
495
+ // Split the list item on soft break into a paragraph node
496
+ activeNode = {
497
+ type: 'paragraph',
498
+ text: '',
499
+ children: [],
500
+ metadata: {
501
+ indentation: lvl,
502
+ alignment: pNode.metadata?.alignment || 'left'
503
+ }
504
+ };
505
+ if (config.includeRawContent) {
506
+ activeNode.rawContent = (0, xmlUtils_js_1.getRawContent)(p, xmlContentString, config);
495
507
  }
508
+ nodes.push(activeNode);
509
+ }
510
+ else {
511
+ // In a normal paragraph, just add a newline
512
+ activeNode.text += "\n";
513
+ activeNode.children?.push({ type: 'text', text: "\n" });
496
514
  }
497
- pNode.children?.push(textNode);
498
515
  }
499
516
  }
500
- if (pNode.text) {
501
- nodes.push(pNode);
502
- }
503
517
  }
504
518
  }
505
- return nodes;
519
+ return nodes.filter(n => n.text?.trim() || (n.children && n.children.length > 0));
506
520
  };
507
521
  /**
508
522
  * Recursively traverses a PowerPoint shape tree (p:spTree),
@@ -692,10 +706,10 @@ const parsePowerPoint = async (buffer, config) => {
692
706
  if (config.ocr) {
693
707
  if (attachment.mimeType.startsWith('image/')) {
694
708
  try {
695
- attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(media.content, { language: config.ocrLanguage, ...config.ocrConfig })).trim();
709
+ attachment.ocrText = (await (0, ocrUtils_js_1.performOcr)(media.content, { ...config.ocrConfig })).trim();
696
710
  }
697
711
  catch (e) {
698
- (0, errorUtils_js_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
712
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.OCR_FAILED, config, attachment.name, e);
699
713
  }
700
714
  }
701
715
  }
@@ -717,7 +731,7 @@ const parsePowerPoint = async (buffer, config) => {
717
731
  attachment.chartData = chartData;
718
732
  }
719
733
  catch (e) {
720
- (0, errorUtils_js_1.logWarning)(`Failed to extract text from chart ${chart.path}:`, config, e);
734
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.CHART_DATA_EXTRACTION_FAILED, config, chart.path, e);
721
735
  }
722
736
  }
723
737
  // Loop through nodes to find images and charts and link their text and chartData
@@ -733,7 +747,7 @@ const parsePowerPoint = async (buffer, config) => {
733
747
  node.text = attachment.ocrText;
734
748
  }
735
749
  if (node.type === 'chart') {
736
- node.text = attachment.chartData?.rawTexts.join(config.newlineDelimiter || '\n');
750
+ node.text = attachment.chartData?.rawTexts.join(config.newlineDelimiter);
737
751
  }
738
752
  }
739
753
  }
@@ -752,24 +766,19 @@ const parsePowerPoint = async (buffer, config) => {
752
766
  return aIsNote - bIsNote;
753
767
  });
754
768
  }
755
- return {
756
- type: 'pptx',
757
- metadata: metadata,
758
- content: content,
759
- attachments: attachments,
760
- toText: () => content.map(c => {
761
- // Recursive text extraction
762
- const getText = (node) => {
763
- let t = '';
764
- if (node.children) {
765
- t += node.children.map(getText).filter(t => t != '').join(!node.children[0]?.children ? '' : config.newlineDelimiter ?? '\n');
766
- }
767
- else
768
- t += node.text || '';
769
- return t;
770
- };
771
- return getText(c);
772
- }).filter(t => t != '').join(config.newlineDelimiter ?? '\n')
773
- };
769
+ const toTextSync = () => content.map(c => {
770
+ // Recursive text extraction
771
+ const getText = (node) => {
772
+ let t = '';
773
+ if (node.children) {
774
+ t += node.children.map(getText).filter(t => t != '').join(!node.children[0]?.children ? '' : config.newlineDelimiter);
775
+ }
776
+ else
777
+ t += node.text || '';
778
+ return t;
779
+ };
780
+ return getText(c);
781
+ }).filter(t => t != '').join(config.newlineDelimiter);
782
+ return (0, astUtils_js_1.createAST)('pptx', metadata, content, attachments, config, toTextSync);
774
783
  };
775
784
  exports.parsePowerPoint = parsePowerPoint;
@@ -39,7 +39,7 @@
39
39
  * @see https://www.biblioscape.com/rtf15_spec.htm RTF 1.5 Specification
40
40
  * @see https://latex2rtf.sourceforge.net/RTF-Spec-1.2.pdf RTF 1.2 Specification
41
41
  */
42
- import { OfficeParserAST, OfficeParserConfig } from '../types.js';
42
+ import { FullOfficeParserConfig, OfficeParserAST } from '../types.js';
43
43
  /**
44
44
  * Represents an RTF group (content enclosed in braces).
45
45
  * Groups create formatting scopes and can contain other groups, control words, or text.
@@ -130,6 +130,12 @@ export declare class SimpleRtfParser {
130
130
  private index;
131
131
  /** The RTF content as a Buffer */
132
132
  private buffer;
133
+ /** Current code page for character decoding (default is Windows-1252) */
134
+ private codePage;
135
+ /** Cached TextDecoders for different code pages */
136
+ private decoders;
137
+ /** Buffer for consecutive text bytes to handle multi-byte encodings and UTF-8 detection */
138
+ private pendingBytes;
133
139
  /** Total length of the buffer */
134
140
  private length;
135
141
  /**
@@ -140,6 +146,20 @@ export declare class SimpleRtfParser {
140
146
  parse(): RtfGroup;
141
147
  private parseControl;
142
148
  private parseText;
149
+ /**
150
+ * Flushes the pending bytes buffer as a text node to the current group.
151
+ * @param group The group to append the text node to
152
+ */
153
+ private flushPendingText;
154
+ /**
155
+ * Decodes a byte array using a "UTF-8 first" strategy.
156
+ * If the bytes form valid UTF-8 and contain non-ASCII characters, UTF-8 is preferred.
157
+ * Otherwise, falls back to the specified code page.
158
+ * @param bytes The bytes to decode
159
+ * @param codePage The RTF code page ID
160
+ * @returns The decoded string
161
+ */
162
+ private decodeBytes;
143
163
  }
144
164
  /**
145
165
  * Parses an RTF file and returns the AST.
@@ -164,4 +184,4 @@ export declare class SimpleRtfParser {
164
184
  * @param config The parser configuration.
165
185
  * @returns The parsed AST.
166
186
  */
167
- export declare const parseRtf: (buffer: Buffer, config: OfficeParserConfig) => Promise<OfficeParserAST>;
187
+ export declare const parseRtf: (buffer: Buffer, config: FullOfficeParserConfig) => Promise<OfficeParserAST>;