officeparser 6.1.0 → 7.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +284 -86
- package/dist/OfficeConverter.d.ts +46 -0
- package/dist/OfficeConverter.js +72 -0
- package/dist/OfficeGenerator.d.ts +19 -0
- package/dist/OfficeGenerator.js +48 -0
- package/dist/OfficeParser.d.ts +6 -0
- package/dist/OfficeParser.js +55 -28
- package/dist/cli.d.ts +3 -1
- package/dist/cli.js +107 -22
- package/dist/defaults.d.ts +41 -0
- package/dist/defaults.js +172 -0
- package/dist/generators/BaseGenerator.d.ts +58 -0
- package/dist/generators/BaseGenerator.js +107 -0
- package/dist/generators/ChunkingGenerator.d.ts +81 -0
- package/dist/generators/ChunkingGenerator.js +683 -0
- package/dist/generators/CsvGenerator.d.ts +30 -0
- package/dist/generators/CsvGenerator.js +233 -0
- package/dist/generators/HtmlGenerator.d.ts +37 -0
- package/dist/generators/HtmlGenerator.js +1013 -0
- package/dist/generators/MarkdownGenerator.d.ts +59 -0
- package/dist/generators/MarkdownGenerator.js +481 -0
- package/dist/generators/PdfGenerator.d.ts +22 -0
- package/dist/generators/PdfGenerator.js +118 -0
- package/dist/generators/RtfGenerator.d.ts +15 -0
- package/dist/generators/RtfGenerator.js +208 -0
- package/dist/generators/TextGenerator.d.ts +13 -0
- package/dist/generators/TextGenerator.js +108 -0
- package/dist/index.d.ts +11 -3
- package/dist/index.js +17 -2
- package/dist/index.mjs +2 -2
- package/dist/officeparser.browser.d.ts +878 -5
- package/dist/officeparser.browser.iife.js +703 -49
- package/dist/officeparser.browser.mjs +703 -49
- package/dist/parsers/CsvParser.d.ts +9 -0
- package/dist/parsers/CsvParser.js +110 -0
- package/dist/parsers/ExcelParser.d.ts +2 -2
- package/dist/parsers/ExcelParser.js +145 -114
- package/dist/parsers/HtmlParser.d.ts +2 -0
- package/dist/parsers/HtmlParser.js +539 -0
- package/dist/parsers/MarkdownParser.d.ts +2 -0
- package/dist/parsers/MarkdownParser.js +360 -0
- package/dist/parsers/OpenOfficeParser.d.ts +2 -2
- package/dist/parsers/OpenOfficeParser.js +237 -128
- package/dist/parsers/PdfParser.d.ts +2 -2
- package/dist/parsers/PdfParser.js +52 -49
- package/dist/parsers/PowerPointParser.d.ts +2 -2
- package/dist/parsers/PowerPointParser.js +132 -123
- package/dist/parsers/RtfParser.d.ts +22 -2
- package/dist/parsers/RtfParser.js +1398 -1282
- package/dist/parsers/WordParser.d.ts +3 -2
- package/dist/parsers/WordParser.js +333 -115
- package/dist/sbom.cdx.json +103 -103
- package/dist/types.d.ts +833 -5
- package/dist/types.js +71 -0
- package/dist/utils/astUtils.d.ts +16 -0
- package/dist/utils/astUtils.js +32 -0
- package/dist/utils/configUtils.d.ts +26 -0
- package/dist/utils/configUtils.js +140 -0
- package/dist/utils/envUtils.js +56 -2
- package/dist/utils/errorUtils.d.ts +17 -29
- package/dist/utils/errorUtils.js +109 -52
- package/dist/utils/moduleLoader.js +15 -9
- package/dist/utils/ocrUtils.js +2 -1
- package/dist/utils/sheetUtils.d.ts +7 -0
- package/dist/utils/sheetUtils.js +35 -0
- package/dist/utils/styleMapper.d.ts +36 -0
- package/dist/utils/styleMapper.js +224 -0
- package/dist/utils/xmlUtils.d.ts +0 -8
- package/dist/utils/xmlUtils.js +2 -1
- package/package.json +28 -9
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
import { FullOfficeParserConfig, OfficeParserAST } from '../types.js';
|
|
2
|
+
/**
|
|
3
|
+
* Parses a CSV file and extracts a single sheet with rows and cells.
|
|
4
|
+
*
|
|
5
|
+
* @param buffer - The CSV file as a Buffer
|
|
6
|
+
* @param config - Parser configuration
|
|
7
|
+
* @returns A promise resolving to the parsed AST
|
|
8
|
+
*/
|
|
9
|
+
export declare const parseCsv: (buffer: Buffer, config: FullOfficeParserConfig) => Promise<OfficeParserAST>;
|
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.parseCsv = void 0;
|
|
4
|
+
const astUtils_js_1 = require("../utils/astUtils.js");
|
|
5
|
+
/**
|
|
6
|
+
* Parses a CSV file and extracts a single sheet with rows and cells.
|
|
7
|
+
*
|
|
8
|
+
* @param buffer - The CSV file as a Buffer
|
|
9
|
+
* @param config - Parser configuration
|
|
10
|
+
* @returns A promise resolving to the parsed AST
|
|
11
|
+
*/
|
|
12
|
+
const parseCsv = async (buffer, config) => {
|
|
13
|
+
const textStr = buffer.toString('utf-8');
|
|
14
|
+
const delimiter = config.csvDelimiter;
|
|
15
|
+
const records = [];
|
|
16
|
+
let currentRow = [];
|
|
17
|
+
let currentCell = '';
|
|
18
|
+
let inQuotes = false;
|
|
19
|
+
for (let i = 0; i < textStr.length; i++) {
|
|
20
|
+
const char = textStr[i];
|
|
21
|
+
const nextChar = textStr[i + 1];
|
|
22
|
+
if (inQuotes) {
|
|
23
|
+
if (char === '"') {
|
|
24
|
+
if (nextChar === '"') {
|
|
25
|
+
currentCell += '"';
|
|
26
|
+
i++; // Skip the escaped quote
|
|
27
|
+
}
|
|
28
|
+
else {
|
|
29
|
+
inQuotes = false;
|
|
30
|
+
}
|
|
31
|
+
}
|
|
32
|
+
else {
|
|
33
|
+
currentCell += char;
|
|
34
|
+
}
|
|
35
|
+
}
|
|
36
|
+
else {
|
|
37
|
+
if (char === '"') {
|
|
38
|
+
inQuotes = true;
|
|
39
|
+
}
|
|
40
|
+
else if (textStr.substring(i, i + delimiter.length) === delimiter) {
|
|
41
|
+
currentRow.push(currentCell);
|
|
42
|
+
currentCell = '';
|
|
43
|
+
i += delimiter.length - 1; // Skip the rest of the delimiter
|
|
44
|
+
}
|
|
45
|
+
else if (char === '\n') {
|
|
46
|
+
currentRow.push(currentCell);
|
|
47
|
+
records.push(currentRow);
|
|
48
|
+
currentRow = [];
|
|
49
|
+
currentCell = '';
|
|
50
|
+
}
|
|
51
|
+
else if (char === '\r') {
|
|
52
|
+
// Ignore carriage return outside of quotes
|
|
53
|
+
}
|
|
54
|
+
else {
|
|
55
|
+
currentCell += char;
|
|
56
|
+
}
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
if (currentCell !== '' || currentRow.length > 0) {
|
|
60
|
+
currentRow.push(currentCell);
|
|
61
|
+
records.push(currentRow);
|
|
62
|
+
}
|
|
63
|
+
// Filter out trailing empty row if the file ended with a newline
|
|
64
|
+
if (records.length > 0 && records[records.length - 1].length === 1 && records[records.length - 1][0] === '') {
|
|
65
|
+
records.pop();
|
|
66
|
+
}
|
|
67
|
+
const rows = [];
|
|
68
|
+
records.forEach((record, rowIndex) => {
|
|
69
|
+
// Handle comment rows
|
|
70
|
+
if (record.length === 1 && record[0].startsWith('#')) {
|
|
71
|
+
rows.push({
|
|
72
|
+
type: 'comment',
|
|
73
|
+
text: record[0]
|
|
74
|
+
});
|
|
75
|
+
return;
|
|
76
|
+
}
|
|
77
|
+
const cells = [];
|
|
78
|
+
record.forEach((val, colIndex) => {
|
|
79
|
+
if (val && val.trim() !== '') {
|
|
80
|
+
const cellMeta = { row: rowIndex, col: colIndex };
|
|
81
|
+
cells.push({
|
|
82
|
+
type: 'cell',
|
|
83
|
+
text: val,
|
|
84
|
+
metadata: cellMeta,
|
|
85
|
+
children: [{ type: 'text', text: val }]
|
|
86
|
+
});
|
|
87
|
+
}
|
|
88
|
+
});
|
|
89
|
+
if (cells.length > 0) {
|
|
90
|
+
rows.push({
|
|
91
|
+
type: 'row',
|
|
92
|
+
children: cells
|
|
93
|
+
});
|
|
94
|
+
}
|
|
95
|
+
});
|
|
96
|
+
const sheetMeta = { sheetName: 'Sheet1' };
|
|
97
|
+
const sheetNode = {
|
|
98
|
+
type: 'sheet',
|
|
99
|
+
metadata: sheetMeta,
|
|
100
|
+
children: rows,
|
|
101
|
+
rawContent: config.includeRawContent ? textStr : undefined
|
|
102
|
+
};
|
|
103
|
+
const toTextSync = () => {
|
|
104
|
+
return records.map((record) => record.filter(cell => cell.trim() !== '').join(config.newlineDelimiter))
|
|
105
|
+
.join(config.newlineDelimiter)
|
|
106
|
+
.replace(/\n{3,}/g, '\n\n');
|
|
107
|
+
};
|
|
108
|
+
return (0, astUtils_js_1.createAST)('csv', { title: 'Sheet1' }, [sheetNode], [], config, toTextSync);
|
|
109
|
+
};
|
|
110
|
+
exports.parseCsv = parseCsv;
|
|
@@ -21,7 +21,7 @@
|
|
|
21
21
|
* @module ExcelParser
|
|
22
22
|
* @see https://www.ecma-international.org/publications-and-standards/standards/ecma-376/
|
|
23
23
|
*/
|
|
24
|
-
import {
|
|
24
|
+
import { FullOfficeParserConfig, OfficeParserAST } from '../types.js';
|
|
25
25
|
/**
|
|
26
26
|
* Parses an Excel spreadsheet (.xlsx) and extracts sheets, rows, and cells.
|
|
27
27
|
*
|
|
@@ -29,4 +29,4 @@ import { OfficeParserAST, OfficeParserConfig } from '../types.js';
|
|
|
29
29
|
* @param config - Parser configuration
|
|
30
30
|
* @returns A promise resolving to the parsed AST
|
|
31
31
|
*/
|
|
32
|
-
export declare const parseExcel: (buffer: Buffer, config:
|
|
32
|
+
export declare const parseExcel: (buffer: Buffer, config: FullOfficeParserConfig) => Promise<OfficeParserAST>;
|
|
@@ -24,6 +24,8 @@
|
|
|
24
24
|
*/
|
|
25
25
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
26
26
|
exports.parseExcel = void 0;
|
|
27
|
+
const types_js_1 = require("../types.js");
|
|
28
|
+
const astUtils_js_1 = require("../utils/astUtils.js");
|
|
27
29
|
const chartUtils_js_1 = require("../utils/chartUtils.js");
|
|
28
30
|
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
29
31
|
const imageUtils_js_1 = require("../utils/imageUtils.js");
|
|
@@ -305,13 +307,13 @@ const parseExcel = async (buffer, config) => {
|
|
|
305
307
|
if (config.ocr) {
|
|
306
308
|
if (attachment.mimeType.startsWith('image/')) {
|
|
307
309
|
try {
|
|
308
|
-
const ocrText = (await (0, ocrUtils_js_1.performOcr)(media.content, {
|
|
310
|
+
const ocrText = (await (0, ocrUtils_js_1.performOcr)(media.content, { ...config.ocrConfig })).trim();
|
|
309
311
|
if (ocrText) {
|
|
310
312
|
attachment.ocrText = ocrText;
|
|
311
313
|
}
|
|
312
314
|
}
|
|
313
315
|
catch (e) {
|
|
314
|
-
(0, errorUtils_js_1.logWarning)(
|
|
316
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.OCR_FAILED, config, attachment.name, e);
|
|
315
317
|
}
|
|
316
318
|
}
|
|
317
319
|
}
|
|
@@ -330,7 +332,7 @@ const parseExcel = async (buffer, config) => {
|
|
|
330
332
|
attachment.chartData = chartData;
|
|
331
333
|
}
|
|
332
334
|
catch (e) {
|
|
333
|
-
(0, errorUtils_js_1.logWarning)(
|
|
335
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.CHART_DATA_EXTRACTION_FAILED, config, chart.path, e);
|
|
334
336
|
}
|
|
335
337
|
attachments.push(attachment);
|
|
336
338
|
}
|
|
@@ -410,104 +412,138 @@ const parseExcel = async (buffer, config) => {
|
|
|
410
412
|
rawContents.push(file.content.toString());
|
|
411
413
|
}
|
|
412
414
|
const rows = [];
|
|
413
|
-
const
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
|
|
421
|
-
|
|
422
|
-
|
|
423
|
-
|
|
424
|
-
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
|
|
429
|
-
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
|
|
441
|
-
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
|
|
446
|
-
|
|
447
|
-
|
|
448
|
-
|
|
449
|
-
|
|
450
|
-
|
|
451
|
-
|
|
452
|
-
|
|
453
|
-
|
|
454
|
-
|
|
455
|
-
|
|
456
|
-
|
|
457
|
-
|
|
458
|
-
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
|
|
462
|
-
|
|
463
|
-
|
|
464
|
-
|
|
465
|
-
|
|
466
|
-
|
|
467
|
-
|
|
468
|
-
|
|
469
|
-
|
|
470
|
-
|
|
471
|
-
// Font defaults from cell style if not in run
|
|
472
|
-
if (!node.formatting.font && cellFormatting.font)
|
|
473
|
-
node.formatting.font = cellFormatting.font;
|
|
474
|
-
if (!node.formatting.size && cellFormatting.size)
|
|
475
|
-
node.formatting.size = cellFormatting.size;
|
|
476
|
-
}
|
|
477
|
-
}
|
|
478
|
-
else {
|
|
479
|
-
// Simple text node
|
|
480
|
-
cellNodes.push({
|
|
481
|
-
type: 'text',
|
|
482
|
-
text: text,
|
|
483
|
-
formatting: cellFormatting
|
|
484
|
-
});
|
|
485
|
-
}
|
|
486
|
-
const cellNode = {
|
|
487
|
-
type: 'cell',
|
|
488
|
-
text: text,
|
|
489
|
-
children: cellNodes,
|
|
490
|
-
metadata: { row: rowIndex, col: colIndex }
|
|
491
|
-
};
|
|
492
|
-
if (config.includeRawContent) {
|
|
493
|
-
cellNode.rawContent = cXml;
|
|
494
|
-
}
|
|
495
|
-
cells.push(cellNode);
|
|
496
|
-
}
|
|
415
|
+
const sheetXml = file.content.toString();
|
|
416
|
+
// regex to match <row> elements, capturing:
|
|
417
|
+
// 1. attributes (e.g., r="1")
|
|
418
|
+
// 2. whether it's self-closing (/>)
|
|
419
|
+
// 3. inner content (for non-self-closing rows)
|
|
420
|
+
const rowRegex = /<row\b([^>]*?)(?:(\/>)|(>([\s\S]*?)<\/row>))/g;
|
|
421
|
+
// matchAll provides an iterator over all matches, which is much more efficient than
|
|
422
|
+
// iterating over a massive sparse row range declared in spreadsheet dimensions.
|
|
423
|
+
const rowMatches = sheetXml.matchAll(rowRegex);
|
|
424
|
+
/** Helper to convert Excel column string (A, B, AA, etc.) to 0-based index */
|
|
425
|
+
const colToNumber = (col) => {
|
|
426
|
+
let num = 0;
|
|
427
|
+
for (let i = 0; i < col.length; i++) {
|
|
428
|
+
num = num * 26 + (col.charCodeAt(i) - 'A'.charCodeAt(0) + 1);
|
|
429
|
+
}
|
|
430
|
+
return num - 1;
|
|
431
|
+
};
|
|
432
|
+
let lastRowIndex = -1;
|
|
433
|
+
for (const rowMatch of rowMatches) {
|
|
434
|
+
const rowXml = rowMatch[0];
|
|
435
|
+
const rowAttrs = rowMatch[1];
|
|
436
|
+
const isSelfClosing = !!rowMatch[2];
|
|
437
|
+
const rowContent = rowMatch[4] || "";
|
|
438
|
+
if (!isSelfClosing && !rowContent.includes('<c'))
|
|
439
|
+
continue;
|
|
440
|
+
const cells = [];
|
|
441
|
+
// regex to match <c> (cell) elements within a row, capturing:
|
|
442
|
+
// 1. cell attributes (e.g., r="A1", t="s")
|
|
443
|
+
// 2. whether it's self-closing (/>)
|
|
444
|
+
// 3. inner content (e.g., <v> value)
|
|
445
|
+
const cRegex = /<c\b([^>]*?)(?:(\/>)|(>([\s\S]*?)<\/c>))/g;
|
|
446
|
+
const cMatches = rowContent.matchAll(cRegex);
|
|
447
|
+
const rMatch = rowAttrs.match(/r="(\d+)"/);
|
|
448
|
+
const rowIndex = rMatch ? parseInt(rMatch[1]) - 1 : lastRowIndex + 1;
|
|
449
|
+
lastRowIndex = rowIndex;
|
|
450
|
+
let lastColIndex = -1;
|
|
451
|
+
for (const cMatch of cMatches) {
|
|
452
|
+
const cXml = cMatch[0];
|
|
453
|
+
const cAttrs = cMatch[1];
|
|
454
|
+
const cContent = cMatch[4] || "";
|
|
455
|
+
// Extract cell value
|
|
456
|
+
const typeMatch = cAttrs.match(/t="([a-zA-Z]+)"/);
|
|
457
|
+
const type = typeMatch ? typeMatch[1] : 'n'; // n = number (default)
|
|
458
|
+
const vMatch = cContent.match(/<v>([\s\S]*?)<\/v>/);
|
|
459
|
+
const tMatch = cContent.match(/<t>([\s\S]*?)<\/t>/);
|
|
460
|
+
let text = '';
|
|
461
|
+
let cellNodes = [];
|
|
462
|
+
if (type === 's' && vMatch) {
|
|
463
|
+
const idx = parseInt(vMatch[1]);
|
|
464
|
+
const content = sharedStrings[idx];
|
|
465
|
+
if (Array.isArray(content)) {
|
|
466
|
+
// Rich text runs
|
|
467
|
+
// Deep copy runs to avoid reference issues if reused
|
|
468
|
+
cellNodes = JSON.parse(JSON.stringify(content));
|
|
469
|
+
text = cellNodes.map(n => n.text).join('');
|
|
470
|
+
}
|
|
471
|
+
else {
|
|
472
|
+
text = content || '';
|
|
497
473
|
}
|
|
498
474
|
}
|
|
499
|
-
if (
|
|
500
|
-
|
|
501
|
-
|
|
502
|
-
|
|
503
|
-
|
|
475
|
+
else if (type === 'inlineStr' && tMatch) {
|
|
476
|
+
text = tMatch[1].trim();
|
|
477
|
+
}
|
|
478
|
+
else if (vMatch) {
|
|
479
|
+
text = vMatch[1].trim();
|
|
480
|
+
}
|
|
481
|
+
// Parse cell coordinate
|
|
482
|
+
const coordMatch = cAttrs.match(/r="([A-Z]+)(\d+)"/);
|
|
483
|
+
let colIndex;
|
|
484
|
+
if (coordMatch) {
|
|
485
|
+
colIndex = colToNumber(coordMatch[1]);
|
|
486
|
+
// If row index is missing in cell coord (unlikely but possible), use rowIndex
|
|
487
|
+
}
|
|
488
|
+
else {
|
|
489
|
+
colIndex = lastColIndex + 1;
|
|
490
|
+
}
|
|
491
|
+
lastColIndex = colIndex;
|
|
492
|
+
if (text || cellNodes.length > 0) {
|
|
493
|
+
// Extract cell style index
|
|
494
|
+
const styleMatch = cAttrs.match(/s="(\d+)"/);
|
|
495
|
+
const styleIdx = styleMatch ? parseInt(styleMatch[1]) : undefined;
|
|
496
|
+
const cellFormatting = (styleIdx !== undefined && cellFormatMap[styleIdx]) ? cellFormatMap[styleIdx] : {};
|
|
497
|
+
if (cellNodes.length > 0) {
|
|
498
|
+
// If we have specific runs, merge cell styles into them if run style is missing
|
|
499
|
+
// But usually run style overrides cell style (except maybe background)
|
|
500
|
+
for (const node of cellNodes) {
|
|
501
|
+
if (!node.formatting)
|
|
502
|
+
node.formatting = {};
|
|
503
|
+
// Cell background always applies
|
|
504
|
+
if (cellFormatting.backgroundColor)
|
|
505
|
+
node.formatting.backgroundColor = cellFormatting.backgroundColor;
|
|
506
|
+
// Cell alignment always applies
|
|
507
|
+
if (cellFormatting.alignment)
|
|
508
|
+
node.formatting.alignment = cellFormatting.alignment;
|
|
509
|
+
// Font defaults from cell style if not in run
|
|
510
|
+
if (!node.formatting.font && cellFormatting.font)
|
|
511
|
+
node.formatting.font = cellFormatting.font;
|
|
512
|
+
if (!node.formatting.size && cellFormatting.size)
|
|
513
|
+
node.formatting.size = cellFormatting.size;
|
|
514
|
+
}
|
|
515
|
+
}
|
|
516
|
+
else {
|
|
517
|
+
// Simple text node
|
|
518
|
+
cellNodes.push({
|
|
519
|
+
type: 'text',
|
|
520
|
+
text: text,
|
|
521
|
+
formatting: cellFormatting
|
|
522
|
+
});
|
|
523
|
+
}
|
|
524
|
+
const cellNode = {
|
|
525
|
+
type: 'cell',
|
|
526
|
+
text: text,
|
|
527
|
+
children: cellNodes,
|
|
528
|
+
metadata: { row: rowIndex, col: colIndex }
|
|
504
529
|
};
|
|
505
530
|
if (config.includeRawContent) {
|
|
506
|
-
|
|
531
|
+
cellNode.rawContent = cXml;
|
|
507
532
|
}
|
|
508
|
-
|
|
533
|
+
cells.push(cellNode);
|
|
509
534
|
}
|
|
510
535
|
}
|
|
536
|
+
if (cells.length > 0) {
|
|
537
|
+
const rowNode = {
|
|
538
|
+
type: 'row',
|
|
539
|
+
children: cells,
|
|
540
|
+
metadata: undefined
|
|
541
|
+
};
|
|
542
|
+
if (config.includeRawContent) {
|
|
543
|
+
rowNode.rawContent = rowXml;
|
|
544
|
+
}
|
|
545
|
+
rows.push(rowNode);
|
|
546
|
+
}
|
|
511
547
|
}
|
|
512
548
|
// Handle Drawings in Sheet (images and charts)
|
|
513
549
|
if (config.extractAttachments) {
|
|
@@ -617,7 +653,7 @@ const parseExcel = async (buffer, config) => {
|
|
|
617
653
|
if (node.type === 'chart') {
|
|
618
654
|
// Link chart data text to chart node
|
|
619
655
|
if (attachment.chartData) {
|
|
620
|
-
node.text = attachment.chartData.rawTexts.join(config.newlineDelimiter
|
|
656
|
+
node.text = attachment.chartData.rawTexts.join(config.newlineDelimiter);
|
|
621
657
|
}
|
|
622
658
|
}
|
|
623
659
|
}
|
|
@@ -628,24 +664,19 @@ const parseExcel = async (buffer, config) => {
|
|
|
628
664
|
}
|
|
629
665
|
};
|
|
630
666
|
assignAttachmentData(content);
|
|
631
|
-
|
|
632
|
-
|
|
633
|
-
|
|
634
|
-
|
|
635
|
-
|
|
636
|
-
|
|
637
|
-
|
|
638
|
-
|
|
639
|
-
|
|
640
|
-
|
|
641
|
-
|
|
642
|
-
|
|
643
|
-
|
|
644
|
-
|
|
645
|
-
return t;
|
|
646
|
-
};
|
|
647
|
-
return getText(c);
|
|
648
|
-
}).filter(t => t != '').join(config.newlineDelimiter ?? '\n')
|
|
649
|
-
};
|
|
667
|
+
const toTextSync = () => content.map(c => {
|
|
668
|
+
// Recursive text extraction
|
|
669
|
+
const getText = (node) => {
|
|
670
|
+
let t = '';
|
|
671
|
+
if (node.children) {
|
|
672
|
+
t += node.children.map(getText).filter(t => t != '').join(!node.children[0]?.children ? '' : config.newlineDelimiter);
|
|
673
|
+
}
|
|
674
|
+
else
|
|
675
|
+
t += node.text || '';
|
|
676
|
+
return t;
|
|
677
|
+
};
|
|
678
|
+
return getText(c);
|
|
679
|
+
}).filter(t => t != '').join(config.newlineDelimiter);
|
|
680
|
+
return (0, astUtils_js_1.createAST)('xlsx', metadata, content, attachments, config, toTextSync);
|
|
650
681
|
};
|
|
651
682
|
exports.parseExcel = parseExcel;
|