officeparser 6.1.0 → 7.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (70) hide show
  1. package/README.md +284 -86
  2. package/dist/OfficeConverter.d.ts +46 -0
  3. package/dist/OfficeConverter.js +72 -0
  4. package/dist/OfficeGenerator.d.ts +19 -0
  5. package/dist/OfficeGenerator.js +48 -0
  6. package/dist/OfficeParser.d.ts +6 -0
  7. package/dist/OfficeParser.js +55 -28
  8. package/dist/cli.d.ts +3 -1
  9. package/dist/cli.js +107 -22
  10. package/dist/defaults.d.ts +41 -0
  11. package/dist/defaults.js +172 -0
  12. package/dist/generators/BaseGenerator.d.ts +58 -0
  13. package/dist/generators/BaseGenerator.js +107 -0
  14. package/dist/generators/ChunkingGenerator.d.ts +81 -0
  15. package/dist/generators/ChunkingGenerator.js +683 -0
  16. package/dist/generators/CsvGenerator.d.ts +30 -0
  17. package/dist/generators/CsvGenerator.js +233 -0
  18. package/dist/generators/HtmlGenerator.d.ts +37 -0
  19. package/dist/generators/HtmlGenerator.js +1013 -0
  20. package/dist/generators/MarkdownGenerator.d.ts +59 -0
  21. package/dist/generators/MarkdownGenerator.js +481 -0
  22. package/dist/generators/PdfGenerator.d.ts +22 -0
  23. package/dist/generators/PdfGenerator.js +118 -0
  24. package/dist/generators/RtfGenerator.d.ts +15 -0
  25. package/dist/generators/RtfGenerator.js +208 -0
  26. package/dist/generators/TextGenerator.d.ts +13 -0
  27. package/dist/generators/TextGenerator.js +108 -0
  28. package/dist/index.d.ts +11 -3
  29. package/dist/index.js +17 -2
  30. package/dist/index.mjs +2 -2
  31. package/dist/officeparser.browser.d.ts +878 -5
  32. package/dist/officeparser.browser.iife.js +703 -49
  33. package/dist/officeparser.browser.mjs +703 -49
  34. package/dist/parsers/CsvParser.d.ts +9 -0
  35. package/dist/parsers/CsvParser.js +110 -0
  36. package/dist/parsers/ExcelParser.d.ts +2 -2
  37. package/dist/parsers/ExcelParser.js +145 -114
  38. package/dist/parsers/HtmlParser.d.ts +2 -0
  39. package/dist/parsers/HtmlParser.js +539 -0
  40. package/dist/parsers/MarkdownParser.d.ts +2 -0
  41. package/dist/parsers/MarkdownParser.js +360 -0
  42. package/dist/parsers/OpenOfficeParser.d.ts +2 -2
  43. package/dist/parsers/OpenOfficeParser.js +237 -128
  44. package/dist/parsers/PdfParser.d.ts +2 -2
  45. package/dist/parsers/PdfParser.js +52 -49
  46. package/dist/parsers/PowerPointParser.d.ts +2 -2
  47. package/dist/parsers/PowerPointParser.js +132 -123
  48. package/dist/parsers/RtfParser.d.ts +22 -2
  49. package/dist/parsers/RtfParser.js +1398 -1282
  50. package/dist/parsers/WordParser.d.ts +3 -2
  51. package/dist/parsers/WordParser.js +333 -115
  52. package/dist/sbom.cdx.json +103 -103
  53. package/dist/types.d.ts +833 -5
  54. package/dist/types.js +71 -0
  55. package/dist/utils/astUtils.d.ts +16 -0
  56. package/dist/utils/astUtils.js +32 -0
  57. package/dist/utils/configUtils.d.ts +26 -0
  58. package/dist/utils/configUtils.js +140 -0
  59. package/dist/utils/envUtils.js +56 -2
  60. package/dist/utils/errorUtils.d.ts +17 -29
  61. package/dist/utils/errorUtils.js +109 -52
  62. package/dist/utils/moduleLoader.js +15 -9
  63. package/dist/utils/ocrUtils.js +2 -1
  64. package/dist/utils/sheetUtils.d.ts +7 -0
  65. package/dist/utils/sheetUtils.js +35 -0
  66. package/dist/utils/styleMapper.d.ts +36 -0
  67. package/dist/utils/styleMapper.js +224 -0
  68. package/dist/utils/xmlUtils.d.ts +0 -8
  69. package/dist/utils/xmlUtils.js +2 -1
  70. package/package.json +28 -9
@@ -0,0 +1,9 @@
1
+ import { FullOfficeParserConfig, OfficeParserAST } from '../types.js';
2
+ /**
3
+ * Parses a CSV file and extracts a single sheet with rows and cells.
4
+ *
5
+ * @param buffer - The CSV file as a Buffer
6
+ * @param config - Parser configuration
7
+ * @returns A promise resolving to the parsed AST
8
+ */
9
+ export declare const parseCsv: (buffer: Buffer, config: FullOfficeParserConfig) => Promise<OfficeParserAST>;
@@ -0,0 +1,110 @@
1
+ "use strict";
2
+ Object.defineProperty(exports, "__esModule", { value: true });
3
+ exports.parseCsv = void 0;
4
+ const astUtils_js_1 = require("../utils/astUtils.js");
5
+ /**
6
+ * Parses a CSV file and extracts a single sheet with rows and cells.
7
+ *
8
+ * @param buffer - The CSV file as a Buffer
9
+ * @param config - Parser configuration
10
+ * @returns A promise resolving to the parsed AST
11
+ */
12
+ const parseCsv = async (buffer, config) => {
13
+ const textStr = buffer.toString('utf-8');
14
+ const delimiter = config.csvDelimiter;
15
+ const records = [];
16
+ let currentRow = [];
17
+ let currentCell = '';
18
+ let inQuotes = false;
19
+ for (let i = 0; i < textStr.length; i++) {
20
+ const char = textStr[i];
21
+ const nextChar = textStr[i + 1];
22
+ if (inQuotes) {
23
+ if (char === '"') {
24
+ if (nextChar === '"') {
25
+ currentCell += '"';
26
+ i++; // Skip the escaped quote
27
+ }
28
+ else {
29
+ inQuotes = false;
30
+ }
31
+ }
32
+ else {
33
+ currentCell += char;
34
+ }
35
+ }
36
+ else {
37
+ if (char === '"') {
38
+ inQuotes = true;
39
+ }
40
+ else if (textStr.substring(i, i + delimiter.length) === delimiter) {
41
+ currentRow.push(currentCell);
42
+ currentCell = '';
43
+ i += delimiter.length - 1; // Skip the rest of the delimiter
44
+ }
45
+ else if (char === '\n') {
46
+ currentRow.push(currentCell);
47
+ records.push(currentRow);
48
+ currentRow = [];
49
+ currentCell = '';
50
+ }
51
+ else if (char === '\r') {
52
+ // Ignore carriage return outside of quotes
53
+ }
54
+ else {
55
+ currentCell += char;
56
+ }
57
+ }
58
+ }
59
+ if (currentCell !== '' || currentRow.length > 0) {
60
+ currentRow.push(currentCell);
61
+ records.push(currentRow);
62
+ }
63
+ // Filter out trailing empty row if the file ended with a newline
64
+ if (records.length > 0 && records[records.length - 1].length === 1 && records[records.length - 1][0] === '') {
65
+ records.pop();
66
+ }
67
+ const rows = [];
68
+ records.forEach((record, rowIndex) => {
69
+ // Handle comment rows
70
+ if (record.length === 1 && record[0].startsWith('#')) {
71
+ rows.push({
72
+ type: 'comment',
73
+ text: record[0]
74
+ });
75
+ return;
76
+ }
77
+ const cells = [];
78
+ record.forEach((val, colIndex) => {
79
+ if (val && val.trim() !== '') {
80
+ const cellMeta = { row: rowIndex, col: colIndex };
81
+ cells.push({
82
+ type: 'cell',
83
+ text: val,
84
+ metadata: cellMeta,
85
+ children: [{ type: 'text', text: val }]
86
+ });
87
+ }
88
+ });
89
+ if (cells.length > 0) {
90
+ rows.push({
91
+ type: 'row',
92
+ children: cells
93
+ });
94
+ }
95
+ });
96
+ const sheetMeta = { sheetName: 'Sheet1' };
97
+ const sheetNode = {
98
+ type: 'sheet',
99
+ metadata: sheetMeta,
100
+ children: rows,
101
+ rawContent: config.includeRawContent ? textStr : undefined
102
+ };
103
+ const toTextSync = () => {
104
+ return records.map((record) => record.filter(cell => cell.trim() !== '').join(config.newlineDelimiter))
105
+ .join(config.newlineDelimiter)
106
+ .replace(/\n{3,}/g, '\n\n');
107
+ };
108
+ return (0, astUtils_js_1.createAST)('csv', { title: 'Sheet1' }, [sheetNode], [], config, toTextSync);
109
+ };
110
+ exports.parseCsv = parseCsv;
@@ -21,7 +21,7 @@
21
21
  * @module ExcelParser
22
22
  * @see https://www.ecma-international.org/publications-and-standards/standards/ecma-376/
23
23
  */
24
- import { OfficeParserAST, OfficeParserConfig } from '../types.js';
24
+ import { FullOfficeParserConfig, OfficeParserAST } from '../types.js';
25
25
  /**
26
26
  * Parses an Excel spreadsheet (.xlsx) and extracts sheets, rows, and cells.
27
27
  *
@@ -29,4 +29,4 @@ import { OfficeParserAST, OfficeParserConfig } from '../types.js';
29
29
  * @param config - Parser configuration
30
30
  * @returns A promise resolving to the parsed AST
31
31
  */
32
- export declare const parseExcel: (buffer: Buffer, config: OfficeParserConfig) => Promise<OfficeParserAST>;
32
+ export declare const parseExcel: (buffer: Buffer, config: FullOfficeParserConfig) => Promise<OfficeParserAST>;
@@ -24,6 +24,8 @@
24
24
  */
25
25
  Object.defineProperty(exports, "__esModule", { value: true });
26
26
  exports.parseExcel = void 0;
27
+ const types_js_1 = require("../types.js");
28
+ const astUtils_js_1 = require("../utils/astUtils.js");
27
29
  const chartUtils_js_1 = require("../utils/chartUtils.js");
28
30
  const errorUtils_js_1 = require("../utils/errorUtils.js");
29
31
  const imageUtils_js_1 = require("../utils/imageUtils.js");
@@ -305,13 +307,13 @@ const parseExcel = async (buffer, config) => {
305
307
  if (config.ocr) {
306
308
  if (attachment.mimeType.startsWith('image/')) {
307
309
  try {
308
- const ocrText = (await (0, ocrUtils_js_1.performOcr)(media.content, { language: config.ocrLanguage, ...config.ocrConfig })).trim();
310
+ const ocrText = (await (0, ocrUtils_js_1.performOcr)(media.content, { ...config.ocrConfig })).trim();
309
311
  if (ocrText) {
310
312
  attachment.ocrText = ocrText;
311
313
  }
312
314
  }
313
315
  catch (e) {
314
- (0, errorUtils_js_1.logWarning)(`OCR failed for ${attachment.name}:`, config, e);
316
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.OCR_FAILED, config, attachment.name, e);
315
317
  }
316
318
  }
317
319
  }
@@ -330,7 +332,7 @@ const parseExcel = async (buffer, config) => {
330
332
  attachment.chartData = chartData;
331
333
  }
332
334
  catch (e) {
333
- (0, errorUtils_js_1.logWarning)(`Failed to extract chart data from ${chart.path}:`, config, e);
335
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.CHART_DATA_EXTRACTION_FAILED, config, chart.path, e);
334
336
  }
335
337
  attachments.push(attachment);
336
338
  }
@@ -410,104 +412,138 @@ const parseExcel = async (buffer, config) => {
410
412
  rawContents.push(file.content.toString());
411
413
  }
412
414
  const rows = [];
413
- const rowRegex = /<row.*?>[\s\S]*?<\/row>/g;
414
- const rowMatches = file.content.toString().match(rowRegex);
415
- if (rowMatches) {
416
- for (const rowXml of rowMatches) {
417
- const cells = [];
418
- const cRegex = /<c.*?>[\s\S]*?<\/c>/g;
419
- const cMatches = rowXml.match(cRegex);
420
- const rMatch = rowXml.match(/r="(\d+)"/);
421
- const rowIndex = rMatch ? parseInt(rMatch[1]) - 1 : 0;
422
- if (cMatches) {
423
- for (const cXml of cMatches) {
424
- // Extract cell value
425
- const typeMatch = cXml.match(/t="([a-z]+)"/);
426
- const type = typeMatch ? typeMatch[1] : 'n'; // n = number (default)
427
- const vMatch = cXml.match(/<v>(.*?)<\/v>/);
428
- const tMatch = cXml.match(/<t>(.*?)<\/t>/);
429
- let text = '';
430
- let cellNodes = [];
431
- if (type === 's' && vMatch) {
432
- const idx = parseInt(vMatch[1]);
433
- const content = sharedStrings[idx];
434
- if (Array.isArray(content)) {
435
- // Rich text runs
436
- // Deep copy runs to avoid reference issues if reused
437
- cellNodes = JSON.parse(JSON.stringify(content));
438
- text = cellNodes.map(n => n.text).join('');
439
- }
440
- else {
441
- text = content || '';
442
- }
443
- }
444
- else if (type === 'inlineStr' && tMatch) {
445
- text = tMatch[1];
446
- }
447
- else if (vMatch) {
448
- text = vMatch[1];
449
- }
450
- // Parse cell coordinate
451
- const coordMatch = cXml.match(/r="([A-Z]+)(\d+)"/);
452
- const colStr = coordMatch ? coordMatch[1] : '';
453
- const colIndex = colStr.charCodeAt(0) - 'A'.charCodeAt(0);
454
- if (text || cellNodes.length > 0) {
455
- // Extract cell style index
456
- const styleMatch = cXml.match(/s="(\d+)"/);
457
- const styleIdx = styleMatch ? parseInt(styleMatch[1]) : undefined;
458
- const cellFormatting = (styleIdx !== undefined && cellFormatMap[styleIdx]) ? cellFormatMap[styleIdx] : {};
459
- if (cellNodes.length > 0) {
460
- // If we have specific runs, merge cell styles into them if run style is missing
461
- // But usually run style overrides cell style (except maybe background)
462
- for (const node of cellNodes) {
463
- if (!node.formatting)
464
- node.formatting = {};
465
- // Cell background always applies
466
- if (cellFormatting.backgroundColor)
467
- node.formatting.backgroundColor = cellFormatting.backgroundColor;
468
- // Cell alignment always applies
469
- if (cellFormatting.alignment)
470
- node.formatting.alignment = cellFormatting.alignment;
471
- // Font defaults from cell style if not in run
472
- if (!node.formatting.font && cellFormatting.font)
473
- node.formatting.font = cellFormatting.font;
474
- if (!node.formatting.size && cellFormatting.size)
475
- node.formatting.size = cellFormatting.size;
476
- }
477
- }
478
- else {
479
- // Simple text node
480
- cellNodes.push({
481
- type: 'text',
482
- text: text,
483
- formatting: cellFormatting
484
- });
485
- }
486
- const cellNode = {
487
- type: 'cell',
488
- text: text,
489
- children: cellNodes,
490
- metadata: { row: rowIndex, col: colIndex }
491
- };
492
- if (config.includeRawContent) {
493
- cellNode.rawContent = cXml;
494
- }
495
- cells.push(cellNode);
496
- }
415
+ const sheetXml = file.content.toString();
416
+ // regex to match <row> elements, capturing:
417
+ // 1. attributes (e.g., r="1")
418
+ // 2. whether it's self-closing (/>)
419
+ // 3. inner content (for non-self-closing rows)
420
+ const rowRegex = /<row\b([^>]*?)(?:(\/>)|(>([\s\S]*?)<\/row>))/g;
421
+ // matchAll provides an iterator over all matches, which is much more efficient than
422
+ // iterating over a massive sparse row range declared in spreadsheet dimensions.
423
+ const rowMatches = sheetXml.matchAll(rowRegex);
424
+ /** Helper to convert Excel column string (A, B, AA, etc.) to 0-based index */
425
+ const colToNumber = (col) => {
426
+ let num = 0;
427
+ for (let i = 0; i < col.length; i++) {
428
+ num = num * 26 + (col.charCodeAt(i) - 'A'.charCodeAt(0) + 1);
429
+ }
430
+ return num - 1;
431
+ };
432
+ let lastRowIndex = -1;
433
+ for (const rowMatch of rowMatches) {
434
+ const rowXml = rowMatch[0];
435
+ const rowAttrs = rowMatch[1];
436
+ const isSelfClosing = !!rowMatch[2];
437
+ const rowContent = rowMatch[4] || "";
438
+ if (!isSelfClosing && !rowContent.includes('<c'))
439
+ continue;
440
+ const cells = [];
441
+ // regex to match <c> (cell) elements within a row, capturing:
442
+ // 1. cell attributes (e.g., r="A1", t="s")
443
+ // 2. whether it's self-closing (/>)
444
+ // 3. inner content (e.g., <v> value)
445
+ const cRegex = /<c\b([^>]*?)(?:(\/>)|(>([\s\S]*?)<\/c>))/g;
446
+ const cMatches = rowContent.matchAll(cRegex);
447
+ const rMatch = rowAttrs.match(/r="(\d+)"/);
448
+ const rowIndex = rMatch ? parseInt(rMatch[1]) - 1 : lastRowIndex + 1;
449
+ lastRowIndex = rowIndex;
450
+ let lastColIndex = -1;
451
+ for (const cMatch of cMatches) {
452
+ const cXml = cMatch[0];
453
+ const cAttrs = cMatch[1];
454
+ const cContent = cMatch[4] || "";
455
+ // Extract cell value
456
+ const typeMatch = cAttrs.match(/t="([a-zA-Z]+)"/);
457
+ const type = typeMatch ? typeMatch[1] : 'n'; // n = number (default)
458
+ const vMatch = cContent.match(/<v>([\s\S]*?)<\/v>/);
459
+ const tMatch = cContent.match(/<t>([\s\S]*?)<\/t>/);
460
+ let text = '';
461
+ let cellNodes = [];
462
+ if (type === 's' && vMatch) {
463
+ const idx = parseInt(vMatch[1]);
464
+ const content = sharedStrings[idx];
465
+ if (Array.isArray(content)) {
466
+ // Rich text runs
467
+ // Deep copy runs to avoid reference issues if reused
468
+ cellNodes = JSON.parse(JSON.stringify(content));
469
+ text = cellNodes.map(n => n.text).join('');
470
+ }
471
+ else {
472
+ text = content || '';
497
473
  }
498
474
  }
499
- if (cells.length > 0) {
500
- const rowNode = {
501
- type: 'row',
502
- children: cells,
503
- metadata: undefined
475
+ else if (type === 'inlineStr' && tMatch) {
476
+ text = tMatch[1].trim();
477
+ }
478
+ else if (vMatch) {
479
+ text = vMatch[1].trim();
480
+ }
481
+ // Parse cell coordinate
482
+ const coordMatch = cAttrs.match(/r="([A-Z]+)(\d+)"/);
483
+ let colIndex;
484
+ if (coordMatch) {
485
+ colIndex = colToNumber(coordMatch[1]);
486
+ // If row index is missing in cell coord (unlikely but possible), use rowIndex
487
+ }
488
+ else {
489
+ colIndex = lastColIndex + 1;
490
+ }
491
+ lastColIndex = colIndex;
492
+ if (text || cellNodes.length > 0) {
493
+ // Extract cell style index
494
+ const styleMatch = cAttrs.match(/s="(\d+)"/);
495
+ const styleIdx = styleMatch ? parseInt(styleMatch[1]) : undefined;
496
+ const cellFormatting = (styleIdx !== undefined && cellFormatMap[styleIdx]) ? cellFormatMap[styleIdx] : {};
497
+ if (cellNodes.length > 0) {
498
+ // If we have specific runs, merge cell styles into them if run style is missing
499
+ // But usually run style overrides cell style (except maybe background)
500
+ for (const node of cellNodes) {
501
+ if (!node.formatting)
502
+ node.formatting = {};
503
+ // Cell background always applies
504
+ if (cellFormatting.backgroundColor)
505
+ node.formatting.backgroundColor = cellFormatting.backgroundColor;
506
+ // Cell alignment always applies
507
+ if (cellFormatting.alignment)
508
+ node.formatting.alignment = cellFormatting.alignment;
509
+ // Font defaults from cell style if not in run
510
+ if (!node.formatting.font && cellFormatting.font)
511
+ node.formatting.font = cellFormatting.font;
512
+ if (!node.formatting.size && cellFormatting.size)
513
+ node.formatting.size = cellFormatting.size;
514
+ }
515
+ }
516
+ else {
517
+ // Simple text node
518
+ cellNodes.push({
519
+ type: 'text',
520
+ text: text,
521
+ formatting: cellFormatting
522
+ });
523
+ }
524
+ const cellNode = {
525
+ type: 'cell',
526
+ text: text,
527
+ children: cellNodes,
528
+ metadata: { row: rowIndex, col: colIndex }
504
529
  };
505
530
  if (config.includeRawContent) {
506
- rowNode.rawContent = rowXml;
531
+ cellNode.rawContent = cXml;
507
532
  }
508
- rows.push(rowNode);
533
+ cells.push(cellNode);
509
534
  }
510
535
  }
536
+ if (cells.length > 0) {
537
+ const rowNode = {
538
+ type: 'row',
539
+ children: cells,
540
+ metadata: undefined
541
+ };
542
+ if (config.includeRawContent) {
543
+ rowNode.rawContent = rowXml;
544
+ }
545
+ rows.push(rowNode);
546
+ }
511
547
  }
512
548
  // Handle Drawings in Sheet (images and charts)
513
549
  if (config.extractAttachments) {
@@ -617,7 +653,7 @@ const parseExcel = async (buffer, config) => {
617
653
  if (node.type === 'chart') {
618
654
  // Link chart data text to chart node
619
655
  if (attachment.chartData) {
620
- node.text = attachment.chartData.rawTexts.join(config.newlineDelimiter || '\n');
656
+ node.text = attachment.chartData.rawTexts.join(config.newlineDelimiter);
621
657
  }
622
658
  }
623
659
  }
@@ -628,24 +664,19 @@ const parseExcel = async (buffer, config) => {
628
664
  }
629
665
  };
630
666
  assignAttachmentData(content);
631
- return {
632
- type: 'xlsx',
633
- metadata: metadata,
634
- content: content,
635
- attachments: attachments,
636
- toText: () => content.map(c => {
637
- // Recursive text extraction
638
- const getText = (node) => {
639
- let t = '';
640
- if (node.children) {
641
- t += node.children.map(getText).filter(t => t != '').join(!node.children[0]?.children ? '' : config.newlineDelimiter ?? '\n');
642
- }
643
- else
644
- t += node.text || '';
645
- return t;
646
- };
647
- return getText(c);
648
- }).filter(t => t != '').join(config.newlineDelimiter ?? '\n')
649
- };
667
+ const toTextSync = () => content.map(c => {
668
+ // Recursive text extraction
669
+ const getText = (node) => {
670
+ let t = '';
671
+ if (node.children) {
672
+ t += node.children.map(getText).filter(t => t != '').join(!node.children[0]?.children ? '' : config.newlineDelimiter);
673
+ }
674
+ else
675
+ t += node.text || '';
676
+ return t;
677
+ };
678
+ return getText(c);
679
+ }).filter(t => t != '').join(config.newlineDelimiter);
680
+ return (0, astUtils_js_1.createAST)('xlsx', metadata, content, attachments, config, toTextSync);
650
681
  };
651
682
  exports.parseExcel = parseExcel;
@@ -0,0 +1,2 @@
1
+ import { FullOfficeParserConfig, OfficeParserAST } from '../types.js';
2
+ export declare const parseHtml: (buffer: Buffer, config: FullOfficeParserConfig) => Promise<OfficeParserAST>;