officeparser 6.1.1 → 7.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. package/README.md +301 -26
  2. package/dist/OfficeConverter.d.ts +46 -0
  3. package/dist/OfficeConverter.js +72 -0
  4. package/dist/OfficeGenerator.d.ts +19 -0
  5. package/dist/OfficeGenerator.js +48 -0
  6. package/dist/OfficeParser.d.ts +6 -0
  7. package/dist/OfficeParser.js +74 -31
  8. package/dist/cli.d.ts +3 -1
  9. package/dist/cli.js +106 -22
  10. package/dist/defaults.d.ts +41 -0
  11. package/dist/defaults.js +172 -0
  12. package/dist/generators/BaseGenerator.d.ts +58 -0
  13. package/dist/generators/BaseGenerator.js +107 -0
  14. package/dist/generators/ChunkingGenerator.d.ts +81 -0
  15. package/dist/generators/ChunkingGenerator.js +683 -0
  16. package/dist/generators/CsvGenerator.d.ts +30 -0
  17. package/dist/generators/CsvGenerator.js +233 -0
  18. package/dist/generators/HtmlGenerator.d.ts +37 -0
  19. package/dist/generators/HtmlGenerator.js +1013 -0
  20. package/dist/generators/MarkdownGenerator.d.ts +59 -0
  21. package/dist/generators/MarkdownGenerator.js +481 -0
  22. package/dist/generators/PdfGenerator.d.ts +22 -0
  23. package/dist/generators/PdfGenerator.js +118 -0
  24. package/dist/generators/RtfGenerator.d.ts +15 -0
  25. package/dist/generators/RtfGenerator.js +208 -0
  26. package/dist/generators/TextGenerator.d.ts +13 -0
  27. package/dist/generators/TextGenerator.js +108 -0
  28. package/dist/index.d.ts +11 -3
  29. package/dist/index.js +17 -2
  30. package/dist/index.mjs +2 -2
  31. package/dist/officeparser.browser.d.ts +828 -5
  32. package/dist/officeparser.browser.iife.js +703 -52
  33. package/dist/officeparser.browser.mjs +703 -52
  34. package/dist/parsers/CsvParser.d.ts +9 -0
  35. package/dist/parsers/CsvParser.js +110 -0
  36. package/dist/parsers/ExcelParser.d.ts +2 -2
  37. package/dist/parsers/ExcelParser.js +145 -114
  38. package/dist/parsers/HtmlParser.d.ts +2 -0
  39. package/dist/parsers/HtmlParser.js +539 -0
  40. package/dist/parsers/MarkdownParser.d.ts +2 -0
  41. package/dist/parsers/MarkdownParser.js +360 -0
  42. package/dist/parsers/OpenOfficeParser.d.ts +2 -2
  43. package/dist/parsers/OpenOfficeParser.js +140 -79
  44. package/dist/parsers/PdfParser.d.ts +2 -2
  45. package/dist/parsers/PdfParser.js +52 -49
  46. package/dist/parsers/PowerPointParser.d.ts +2 -2
  47. package/dist/parsers/PowerPointParser.js +20 -23
  48. package/dist/parsers/RtfParser.d.ts +2 -2
  49. package/dist/parsers/RtfParser.js +1291 -1240
  50. package/dist/parsers/WordParser.d.ts +2 -2
  51. package/dist/parsers/WordParser.js +232 -97
  52. package/dist/sbom.cdx.json +99 -99
  53. package/dist/types.d.ts +783 -5
  54. package/dist/types.js +73 -0
  55. package/dist/utils/astUtils.d.ts +16 -0
  56. package/dist/utils/astUtils.js +32 -0
  57. package/dist/utils/configUtils.d.ts +26 -0
  58. package/dist/utils/configUtils.js +140 -0
  59. package/dist/utils/envUtils.d.ts +8 -3
  60. package/dist/utils/envUtils.js +117 -34
  61. package/dist/utils/errorUtils.d.ts +17 -29
  62. package/dist/utils/errorUtils.js +110 -52
  63. package/dist/utils/moduleLoader.js +19 -11
  64. package/dist/utils/ocrUtils.js +2 -1
  65. package/dist/utils/sheetUtils.d.ts +7 -0
  66. package/dist/utils/sheetUtils.js +35 -0
  67. package/dist/utils/styleMapper.d.ts +36 -0
  68. package/dist/utils/styleMapper.js +224 -0
  69. package/dist/utils/xmlUtils.d.ts +0 -8
  70. package/dist/utils/xmlUtils.js +2 -1
  71. package/package.json +26 -7
@@ -0,0 +1,30 @@
1
+ import { ConversionResult, GeneratorConfig, OfficeParserAST } from '../types.js';
2
+ import { BaseGenerator } from './BaseGenerator.js';
3
+ /**
4
+ * Generates CSV files from an AST.
5
+ */
6
+ export declare class CsvGenerator extends BaseGenerator<'csv'> {
7
+ constructor(ast: OfficeParserAST, config?: GeneratorConfig<'csv'>);
8
+ /**
9
+ * Generates CSV content from the provided AST.
10
+ *
11
+ * @returns A CSV string or a ZIP archive containing multiple CSVs
12
+ */
13
+ generate(): Promise<ConversionResult>;
14
+ /**
15
+ * Recursively finds all nodes that can be treated as sheets (sheet or table).
16
+ */
17
+ private collectSheetLikeNodes;
18
+ /**
19
+ * Renders a sheet or table node to raw row data.
20
+ */
21
+ private renderNodeToRows;
22
+ /**
23
+ * Escapes a value for CSV formatting.
24
+ */
25
+ private escapeCsvValue;
26
+ /**
27
+ * Renders metadata as comments.
28
+ */
29
+ private renderMetadata;
30
+ }
@@ -0,0 +1,233 @@
1
+ "use strict";
2
+ Object.defineProperty(exports, "__esModule", { value: true });
3
+ exports.CsvGenerator = void 0;
4
+ const fflate_1 = require("fflate");
5
+ const types_js_1 = require("../types.js");
6
+ const sheetUtils_js_1 = require("../utils/sheetUtils.js");
7
+ const BaseGenerator_js_1 = require("./BaseGenerator.js");
8
+ /**
9
+ * Generates CSV files from an AST.
10
+ */
11
+ class CsvGenerator extends BaseGenerator_js_1.BaseGenerator {
12
+ constructor(ast, config) {
13
+ super('csv', ast, config);
14
+ }
15
+ /**
16
+ * Generates CSV content from the provided AST.
17
+ *
18
+ * @returns A CSV string or a ZIP archive containing multiple CSVs
19
+ */
20
+ async generate() {
21
+ const csvConfig = this.config.csvConfig;
22
+ const delimiter = csvConfig.columnDelimiter;
23
+ const mergeSheets = csvConfig.mergeSheets;
24
+ // 1. Collect all "sheet-like" nodes (sheets and tables)
25
+ const sheetNodes = await this.collectSheetLikeNodes(this.ast.content);
26
+ if (sheetNodes.length === 0) {
27
+ return { value: '', messages: this.messages };
28
+ }
29
+ const metadataHeader = this.config.renderMetadata ? this.renderMetadata(this.ast) : '';
30
+ // 2. Filter sheets based on range
31
+ let selectedNodes = sheetNodes;
32
+ if (csvConfig.sheets) {
33
+ const indices = (0, sheetUtils_js_1.parseRangeString)(csvConfig.sheets);
34
+ selectedNodes = indices
35
+ .filter(i => i > 0 && i <= sheetNodes.length)
36
+ .map(i => sheetNodes[i - 1]);
37
+ }
38
+ if (selectedNodes.length === 0) {
39
+ this.warn(types_js_1.OfficeWarningType.SHEET_RANGE_NOT_FOUND, csvConfig.sheets);
40
+ return { value: '', messages: this.messages };
41
+ }
42
+ // 3. Generate CSV content for each selected node
43
+ const sheetData = [];
44
+ let globalMaxCols = 0;
45
+ for (let i = 0; i < selectedNodes.length; i++) {
46
+ const node = selectedNodes[i];
47
+ const name = node.metadata?.sheetName || `Sheet${i + 1}`;
48
+ const rows = await this.renderNodeToRows(node);
49
+ const maxCols = Math.max(...rows.map(r => r.length), 0);
50
+ if (mergeSheets) {
51
+ globalMaxCols = Math.max(globalMaxCols, maxCols);
52
+ }
53
+ sheetData.push({ name, rows });
54
+ }
55
+ // 4. Handle merging or separate files
56
+ if (mergeSheets) {
57
+ const mergedLines = [];
58
+ if (metadataHeader)
59
+ mergedLines.push(metadataHeader.trim());
60
+ for (const sheet of sheetData) {
61
+ if (sheetData.length > 1)
62
+ mergedLines.push(`# Sheet: ${sheet.name}`);
63
+ for (const row of sheet.rows) {
64
+ const paddedRow = [...row];
65
+ // Don't pad comments
66
+ if (row.length > 1 || (row.length === 1 && !row[0].startsWith('#'))) {
67
+ while (paddedRow.length < globalMaxCols)
68
+ paddedRow.push('');
69
+ }
70
+ mergedLines.push(paddedRow.map(v => this.escapeCsvValue(v, delimiter)).join(delimiter));
71
+ }
72
+ mergedLines.push(''); // Blank line between sheets
73
+ }
74
+ return {
75
+ value: mergedLines.join('\n'),
76
+ messages: this.messages
77
+ };
78
+ }
79
+ else if (sheetData.length > 1) {
80
+ // Create a ZIP archive for multiple sheets
81
+ const zipFiles = {};
82
+ for (const sheet of sheetData) {
83
+ const sheetMaxCols = Math.max(...sheet.rows.map(r => r.length), 0);
84
+ const csvLines = [];
85
+ if (metadataHeader)
86
+ csvLines.push(metadataHeader.trim());
87
+ for (const row of sheet.rows) {
88
+ const paddedRow = [...row];
89
+ // Don't pad comments
90
+ if (row.length > 1 || (row.length === 1 && !row[0].startsWith('#'))) {
91
+ while (paddedRow.length < sheetMaxCols)
92
+ paddedRow.push('');
93
+ }
94
+ csvLines.push(paddedRow.map(v => this.escapeCsvValue(v, delimiter)).join(delimiter));
95
+ }
96
+ const fileName = `${sheet.name.replace(/[^\w\s-]/g, '_')}.csv`;
97
+ zipFiles[fileName] = new TextEncoder().encode(csvLines.join('\n'));
98
+ }
99
+ const zipBuffer = (0, fflate_1.zipSync)(zipFiles);
100
+ return {
101
+ value: zipBuffer,
102
+ messages: this.messages
103
+ };
104
+ }
105
+ else {
106
+ // Single sheet: return as plain string
107
+ const sheet = sheetData[0];
108
+ const sheetMaxCols = Math.max(...sheet.rows.map(r => r.length), 0);
109
+ const csvLines = [];
110
+ if (metadataHeader)
111
+ csvLines.push(metadataHeader.trim());
112
+ for (const row of sheet.rows) {
113
+ const paddedRow = [...row];
114
+ // Don't pad comments
115
+ if (row.length > 1 || (row.length === 1 && !row[0].startsWith('#'))) {
116
+ while (paddedRow.length < sheetMaxCols)
117
+ paddedRow.push('');
118
+ }
119
+ csvLines.push(paddedRow.map(v => this.escapeCsvValue(v, delimiter)).join(delimiter));
120
+ }
121
+ return {
122
+ value: csvLines.join('\n'),
123
+ messages: this.messages
124
+ };
125
+ }
126
+ }
127
+ /**
128
+ * Recursively finds all nodes that can be treated as sheets (sheet or table).
129
+ */
130
+ async collectSheetLikeNodes(nodes) {
131
+ const result = [];
132
+ for (const node of nodes) {
133
+ const override = await this.handleOnNode(node);
134
+ if (override === false) {
135
+ continue;
136
+ }
137
+ if (node.type === 'sheet' || node.type === 'table') {
138
+ result.push(node);
139
+ }
140
+ else if (node.children) {
141
+ result.push(...(await this.collectSheetLikeNodes(node.children)));
142
+ }
143
+ }
144
+ return result;
145
+ }
146
+ /**
147
+ * Renders a sheet or table node to raw row data.
148
+ */
149
+ async renderNodeToRows(node) {
150
+ if (!node.children)
151
+ return [];
152
+ const rows = [];
153
+ const rowNodes = node.children.filter(c => c.type === 'row' || c.type === 'comment');
154
+ // Text processor for cell content
155
+ const cellProcessor = (n, co) => {
156
+ if (n.type === 'text' || n.type === 'code')
157
+ return n.text || '';
158
+ if (n.type === 'break')
159
+ return '\n';
160
+ return co;
161
+ };
162
+ for (const rowNode of rowNodes) {
163
+ const override = await this.handleOnNode(rowNode);
164
+ if (override === false) {
165
+ continue;
166
+ }
167
+ if (typeof override === 'string') {
168
+ rows.push([override]);
169
+ continue;
170
+ }
171
+ if (rowNode.type === 'comment') {
172
+ rows.push([rowNode.text || '']);
173
+ continue;
174
+ }
175
+ if (!rowNode.children) {
176
+ rows.push([]);
177
+ continue;
178
+ }
179
+ const cellNodes = rowNode.children.filter(c => c.type === 'cell');
180
+ const rowValues = [];
181
+ let lastCol = -1;
182
+ for (const cell of cellNodes) {
183
+ const currentCol = cell.metadata?.col ?? (lastCol + 1);
184
+ // Fill gaps with empty strings
185
+ while (lastCol < currentCol - 1) {
186
+ rowValues.push('');
187
+ lastCol++;
188
+ }
189
+ const cellText = await this.processNodeRecursive(cell, cellProcessor);
190
+ rowValues.push(cellText);
191
+ // Handle colSpan: move lastCol forward
192
+ const colSpan = cell.metadata?.colSpan || 1;
193
+ lastCol = currentCol + colSpan - 1;
194
+ }
195
+ rows.push(rowValues);
196
+ }
197
+ return rows;
198
+ }
199
+ /**
200
+ * Escapes a value for CSV formatting.
201
+ */
202
+ escapeCsvValue(val, delimiter) {
203
+ const needsQuotes = val.includes(delimiter) || val.includes('"') || val.includes('\n') || val.includes('\r');
204
+ if (!needsQuotes)
205
+ return val;
206
+ // Double up existing quotes and wrap in quotes
207
+ return `"${val.replace(/"/g, '""')}"`;
208
+ }
209
+ /**
210
+ * Renders metadata as comments.
211
+ */
212
+ renderMetadata(ast) {
213
+ if (!ast.metadata)
214
+ return '';
215
+ const m = ast.metadata;
216
+ let output = '';
217
+ if (m.title)
218
+ output += `# Title: ${m.title}\n`;
219
+ if (m.author)
220
+ output += `# Author: ${m.author}\n`;
221
+ if (m.created)
222
+ output += `# Created: ${new Date(m.created).toLocaleString()}\n`;
223
+ if (m.modified)
224
+ output += `# Modified: ${new Date(m.modified).toLocaleString()}\n`;
225
+ if (m.customProperties) {
226
+ for (const [k, v] of Object.entries(m.customProperties)) {
227
+ output += `# ${k}: ${v}\n`;
228
+ }
229
+ }
230
+ return output ? output + '\n' : '';
231
+ }
232
+ }
233
+ exports.CsvGenerator = CsvGenerator;
@@ -0,0 +1,37 @@
1
+ import { ConversionResult, GeneratorConfig, OfficeContentNode, OfficeParserAST } from '../types.js';
2
+ import { BaseGenerator } from './BaseGenerator.js';
3
+ /**
4
+ * Generates semantic, high-fidelity HTML from an AST.
5
+ */
6
+ export declare class HtmlGenerator extends BaseGenerator<'html'> {
7
+ private chartCounter;
8
+ private isSpreadsheetMode;
9
+ constructor(ast: OfficeParserAST, config?: GeneratorConfig<'html'>);
10
+ /**
11
+ * Generates HTML string from the provided AST.
12
+ *
13
+ * @returns An HTML string
14
+ */
15
+ generate(): Promise<ConversionResult>;
16
+ private renderMetaTags;
17
+ private renderMetadataSummary;
18
+ /**
19
+ * Processes an array of nodes, handling list grouping and nesting.
20
+ */
21
+ private processNodeArray;
22
+ /**
23
+ * Overridden to handle children using processNodeArray for list grouping.
24
+ */
25
+ private tableNestingLevel;
26
+ protected processNodeRecursive(node: OfficeContentNode, processor: (node: OfficeContentNode, childrenOutput: string) => string | Promise<string>, override?: string | boolean | void): Promise<string>;
27
+ /**
28
+ * Internal processor for individual nodes.
29
+ */
30
+ private nodeProcessor;
31
+ private getDefaultTag;
32
+ private formatText;
33
+ private getInlineStyles;
34
+ private getPremiumStyles;
35
+ protected slugify(text: string): string;
36
+ private escape;
37
+ }