officeparser 6.1.1 → 7.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (70) hide show
  1. package/README.md +219 -26
  2. package/dist/OfficeConverter.d.ts +46 -0
  3. package/dist/OfficeConverter.js +72 -0
  4. package/dist/OfficeGenerator.d.ts +19 -0
  5. package/dist/OfficeGenerator.js +48 -0
  6. package/dist/OfficeParser.d.ts +6 -0
  7. package/dist/OfficeParser.js +55 -29
  8. package/dist/cli.d.ts +3 -1
  9. package/dist/cli.js +106 -22
  10. package/dist/defaults.d.ts +41 -0
  11. package/dist/defaults.js +172 -0
  12. package/dist/generators/BaseGenerator.d.ts +58 -0
  13. package/dist/generators/BaseGenerator.js +107 -0
  14. package/dist/generators/ChunkingGenerator.d.ts +81 -0
  15. package/dist/generators/ChunkingGenerator.js +683 -0
  16. package/dist/generators/CsvGenerator.d.ts +30 -0
  17. package/dist/generators/CsvGenerator.js +233 -0
  18. package/dist/generators/HtmlGenerator.d.ts +37 -0
  19. package/dist/generators/HtmlGenerator.js +1013 -0
  20. package/dist/generators/MarkdownGenerator.d.ts +59 -0
  21. package/dist/generators/MarkdownGenerator.js +481 -0
  22. package/dist/generators/PdfGenerator.d.ts +22 -0
  23. package/dist/generators/PdfGenerator.js +118 -0
  24. package/dist/generators/RtfGenerator.d.ts +15 -0
  25. package/dist/generators/RtfGenerator.js +208 -0
  26. package/dist/generators/TextGenerator.d.ts +13 -0
  27. package/dist/generators/TextGenerator.js +108 -0
  28. package/dist/index.d.ts +11 -3
  29. package/dist/index.js +17 -2
  30. package/dist/index.mjs +2 -2
  31. package/dist/officeparser.browser.d.ts +826 -5
  32. package/dist/officeparser.browser.iife.js +703 -52
  33. package/dist/officeparser.browser.mjs +703 -52
  34. package/dist/parsers/CsvParser.d.ts +9 -0
  35. package/dist/parsers/CsvParser.js +110 -0
  36. package/dist/parsers/ExcelParser.d.ts +2 -2
  37. package/dist/parsers/ExcelParser.js +145 -114
  38. package/dist/parsers/HtmlParser.d.ts +2 -0
  39. package/dist/parsers/HtmlParser.js +539 -0
  40. package/dist/parsers/MarkdownParser.d.ts +2 -0
  41. package/dist/parsers/MarkdownParser.js +360 -0
  42. package/dist/parsers/OpenOfficeParser.d.ts +2 -2
  43. package/dist/parsers/OpenOfficeParser.js +140 -79
  44. package/dist/parsers/PdfParser.d.ts +2 -2
  45. package/dist/parsers/PdfParser.js +52 -49
  46. package/dist/parsers/PowerPointParser.d.ts +2 -2
  47. package/dist/parsers/PowerPointParser.js +20 -23
  48. package/dist/parsers/RtfParser.d.ts +2 -2
  49. package/dist/parsers/RtfParser.js +1291 -1240
  50. package/dist/parsers/WordParser.d.ts +2 -2
  51. package/dist/parsers/WordParser.js +232 -97
  52. package/dist/sbom.cdx.json +99 -99
  53. package/dist/types.d.ts +781 -5
  54. package/dist/types.js +71 -0
  55. package/dist/utils/astUtils.d.ts +16 -0
  56. package/dist/utils/astUtils.js +32 -0
  57. package/dist/utils/configUtils.d.ts +26 -0
  58. package/dist/utils/configUtils.js +140 -0
  59. package/dist/utils/envUtils.js +56 -2
  60. package/dist/utils/errorUtils.d.ts +17 -29
  61. package/dist/utils/errorUtils.js +109 -52
  62. package/dist/utils/moduleLoader.js +15 -9
  63. package/dist/utils/ocrUtils.js +2 -1
  64. package/dist/utils/sheetUtils.d.ts +7 -0
  65. package/dist/utils/sheetUtils.js +35 -0
  66. package/dist/utils/styleMapper.d.ts +36 -0
  67. package/dist/utils/styleMapper.js +224 -0
  68. package/dist/utils/xmlUtils.d.ts +0 -8
  69. package/dist/utils/xmlUtils.js +2 -1
  70. package/package.json +27 -8
@@ -0,0 +1,360 @@
1
+ "use strict";
2
+ Object.defineProperty(exports, "__esModule", { value: true });
3
+ exports.parseMarkdown = void 0;
4
+ const astUtils_js_1 = require("../utils/astUtils.js");
5
+ const parseMarkdown = async (buffer, config) => {
6
+ let textStr = buffer.toString('utf-8');
7
+ textStr = textStr.replace(/\r\n/g, '\n');
8
+ const content = [];
9
+ const metadata = {};
10
+ const attachments = [];
11
+ // Parse YAML Front Matter
12
+ if (textStr.startsWith('---\n')) {
13
+ const endIdx = textStr.indexOf('\n---\n', 4);
14
+ if (endIdx !== -1) {
15
+ const frontMatter = textStr.substring(4, endIdx);
16
+ textStr = textStr.substring(endIdx + 5);
17
+ const lines = frontMatter.split('\n');
18
+ const customProps = {};
19
+ for (const line of lines) {
20
+ const match = line.match(/^([^:]+):\s*(.*)$/);
21
+ if (match) {
22
+ const key = match[1].trim();
23
+ let val = match[2].trim().replace(/^"(.*)"$/, '$1');
24
+ if (key === 'title')
25
+ metadata.title = val;
26
+ else if (key === 'author')
27
+ metadata.author = val;
28
+ else if (key === 'created')
29
+ metadata.created = new Date(val);
30
+ else if (key === 'modified')
31
+ metadata.modified = new Date(val);
32
+ else if (key === 'description')
33
+ metadata.description = val;
34
+ else {
35
+ // Try to infer type for custom props
36
+ if (val === 'true')
37
+ customProps[key] = true;
38
+ else if (val === 'false')
39
+ customProps[key] = false;
40
+ else if (!isNaN(Number(val)) && val !== '')
41
+ customProps[key] = Number(val);
42
+ else
43
+ customProps[key] = val;
44
+ }
45
+ }
46
+ }
47
+ if (Object.keys(customProps).length > 0)
48
+ metadata.customProperties = customProps;
49
+ }
50
+ }
51
+ // Extract code blocks first to protect their contents
52
+ const codeBlocks = [];
53
+ textStr = textStr.replace(/^```(\w*)\n([\s\S]*?)\n```/gm, (match, lang, code) => {
54
+ const id = `__CODE_BLOCK_${codeBlocks.length}__`;
55
+ codeBlocks.push(JSON.stringify({ lang, code }));
56
+ return `\n\n${id}\n\n`;
57
+ });
58
+ const parseInline = (text, currentFormatting = {}) => {
59
+ const nodes = [];
60
+ // Regex matches: 1=!, 2=alt, 3=url | 4=bold | 5=italic | 6=strike | 7=code | 8=underline | 9=subscript | 10=superscript
61
+ const regex = /(!?)\[(.*?)\]\((.*?)\)|\*\*(.+?)\*\*|\*(.+?)\*|~~(.+?)~~|`(.+?)`|<u>(.+?)<\/u>|<sub>(.+?)<\/sub>|<sup>(.+?)<\/sup>/g;
62
+ let lastIndex = 0;
63
+ let match;
64
+ while ((match = regex.exec(text)) !== null) {
65
+ if (match.index > lastIndex) {
66
+ nodes.push({ type: 'text', text: text.substring(lastIndex, match.index), formatting: Object.keys(currentFormatting).length > 0 ? { ...currentFormatting } : undefined });
67
+ }
68
+ if (match[2] !== undefined) { // Image or Link
69
+ const isImage = match[1] === '!';
70
+ const altText = match[2];
71
+ const url = match[3];
72
+ if (isImage) {
73
+ if (url.startsWith('data:')) {
74
+ const dataMatch = url.match(/^data:([^;]+);base64,(.*)$/);
75
+ if (dataMatch && config.extractAttachments) {
76
+ const mimeType = dataMatch[1];
77
+ const data = dataMatch[2];
78
+ const name = `image_${attachments.length + 1}.${mimeType.split('/')[1]}`;
79
+ attachments.push({
80
+ type: 'image',
81
+ mimeType,
82
+ data,
83
+ name,
84
+ extension: mimeType.split('/')[1]
85
+ });
86
+ nodes.push({ type: 'image', metadata: { attachmentName: name, altText } });
87
+ }
88
+ else {
89
+ nodes.push({ type: 'image', metadata: { url, altText } });
90
+ }
91
+ }
92
+ else {
93
+ nodes.push({ type: 'image', metadata: { url, altText } });
94
+ }
95
+ }
96
+ else {
97
+ const linkNodes = parseInline(altText, currentFormatting);
98
+ linkNodes.forEach(n => {
99
+ if (n.type === 'text') {
100
+ n.metadata = { link: url, linkType: 'external' };
101
+ }
102
+ });
103
+ nodes.push(...linkNodes);
104
+ }
105
+ }
106
+ else if (match[4]) { // Bold
107
+ nodes.push(...parseInline(match[4], { ...currentFormatting, bold: true }));
108
+ }
109
+ else if (match[5]) { // Italic
110
+ nodes.push(...parseInline(match[5], { ...currentFormatting, italic: true }));
111
+ }
112
+ else if (match[6]) { // Strikethrough
113
+ nodes.push(...parseInline(match[6], { ...currentFormatting, strikethrough: true }));
114
+ }
115
+ else if (match[7]) { // Inline Code
116
+ nodes.push({ type: 'text', text: match[7], formatting: { ...currentFormatting, font: 'monospace' } });
117
+ }
118
+ else if (match[8]) { // Underline
119
+ nodes.push(...parseInline(match[8], { ...currentFormatting, underline: true }));
120
+ }
121
+ else if (match[9]) { // Subscript
122
+ nodes.push(...parseInline(match[9], { ...currentFormatting, subscript: true }));
123
+ }
124
+ else if (match[10]) { // Superscript
125
+ nodes.push(...parseInline(match[10], { ...currentFormatting, superscript: true }));
126
+ }
127
+ lastIndex = regex.lastIndex;
128
+ }
129
+ if (lastIndex < text.length) {
130
+ nodes.push({ type: 'text', text: text.substring(lastIndex), formatting: Object.keys(currentFormatting).length > 0 ? { ...currentFormatting } : undefined });
131
+ }
132
+ return nodes;
133
+ };
134
+ const rawBlocks = textStr.split(/\n\n+/);
135
+ const blocks = [];
136
+ // Sub-split blocks that contain headings or lists without double newlines
137
+ for (const rawBlock of rawBlocks) {
138
+ if (!rawBlock.trim())
139
+ continue;
140
+ // Match headings or lists that might be joined with other text via single newline
141
+ const lines = rawBlock.split('\n');
142
+ let currentSubBlock = [];
143
+ for (const line of lines) {
144
+ const isHeading = !!line.match(/^(?:<a[^>]*><\/a>)*\s*#{1,6}\s+/);
145
+ const isList = !!line.match(/^(\s*)([-*+]|\d+\.)\s+/);
146
+ const isHtmlTag = !!line.match(/^<\/?div[^>]*>$/i);
147
+ const prevWasList = currentSubBlock.length > 0 && !!currentSubBlock[currentSubBlock.length - 1].match(/^(\s*)([-*+]|\d+\.)\s+/);
148
+ // Split if:
149
+ // 1. Current line is a heading
150
+ // 2. Current line is a list item but previous was NOT
151
+ // 3. Current line is NOT a list item but previous WAS
152
+ // 4. Current line is an HTML tag (div)
153
+ if ((isHeading || isHtmlTag || (isList !== prevWasList)) && currentSubBlock.length > 0) {
154
+ blocks.push(currentSubBlock.join('\n'));
155
+ currentSubBlock = [];
156
+ }
157
+ currentSubBlock.push(line);
158
+ // Headings and HTML tags are single-line blocks for our state machine
159
+ if (isHeading || isHtmlTag) {
160
+ blocks.push(currentSubBlock.join('\n'));
161
+ currentSubBlock = [];
162
+ }
163
+ }
164
+ if (currentSubBlock.length > 0) {
165
+ blocks.push(currentSubBlock.join('\n'));
166
+ }
167
+ }
168
+ let listIdCounter = 1;
169
+ let currentAlignment = undefined;
170
+ for (let block of blocks) {
171
+ block = block.trim();
172
+ if (!block)
173
+ continue;
174
+ // Check for alignment wrapper start/end
175
+ const alignStartMatch = block.match(/^<div\s+(?:style="text-align:\s*(left|center|right|justify);?"|align="(left|center|right|justify)")>$/i);
176
+ if (alignStartMatch) {
177
+ currentAlignment = (alignStartMatch[1] || alignStartMatch[2]).toLowerCase();
178
+ continue;
179
+ }
180
+ if (block.match(/^<\/div>$/i)) {
181
+ currentAlignment = undefined;
182
+ continue;
183
+ }
184
+ let alignment = currentAlignment;
185
+ // Check for single-line alignment wrapper (for compatibility)
186
+ const alignMatch = block.match(/^<div\s+(?:style="text-align:\s*(left|center|right|justify);?"|align="(left|center|right|justify)")>\s*([\s\S]*?)\s*<\/div>$/i);
187
+ if (alignMatch) {
188
+ alignment = (alignMatch[1] || alignMatch[2]).toLowerCase();
189
+ block = alignMatch[3];
190
+ }
191
+ // Code Block
192
+ const codeMatch = block.match(/^__CODE_BLOCK_(\d+)__$/);
193
+ if (codeMatch) {
194
+ const data = JSON.parse(codeBlocks[parseInt(codeMatch[1])]);
195
+ content.push({
196
+ type: 'code',
197
+ text: data.code,
198
+ metadata: { language: data.lang }
199
+ });
200
+ continue;
201
+ }
202
+ // Heading (allowing for leading HTML anchors and trailing {#anchor})
203
+ const headingMatch = block.match(/^((?:<a[^>]*><\/a>)*)\s*(#{1,6})\s+(.*?)(?:\s+\{#([^}]+)\})?\s*$/s);
204
+ if (headingMatch) {
205
+ const leadingAnchorsRaw = headingMatch[1];
206
+ const rawText = headingMatch[3];
207
+ const explicitAnchor = headingMatch[4];
208
+ const anchorIds = [];
209
+ if (leadingAnchorsRaw) {
210
+ const idMatches = leadingAnchorsRaw.matchAll(/<a\s+name="([^"]+)"/gi);
211
+ for (const m of idMatches)
212
+ anchorIds.push(m[1]);
213
+ }
214
+ if (explicitAnchor)
215
+ anchorIds.push(explicitAnchor);
216
+ const children = parseInline(rawText);
217
+ content.push({
218
+ type: 'heading',
219
+ text: children.map(c => c.text || '').join(''),
220
+ metadata: {
221
+ level: headingMatch[2].length,
222
+ alignment,
223
+ anchorIds: anchorIds.length > 0 ? anchorIds : undefined
224
+ },
225
+ children
226
+ });
227
+ continue;
228
+ }
229
+ // Blockquote
230
+ const quoteMatch = block.match(/^>\s+(.*)$/s);
231
+ if (quoteMatch) {
232
+ content.push({
233
+ type: 'paragraph',
234
+ metadata: { style: 'Quote' },
235
+ children: parseInline(quoteMatch[1].replace(/^>\s+/gm, ''))
236
+ });
237
+ continue;
238
+ }
239
+ // Lists
240
+ if (block.match(/^(\s*)([-*+]|\d+\.)\s+/)) {
241
+ const lines = block.split('\n');
242
+ const listId = `md-list-${listIdCounter++}`;
243
+ const listCounters = {};
244
+ for (const line of lines) {
245
+ const match = line.match(/^(\s*)([-*+]|\d+\.)\s+(.*)$/);
246
+ if (match) {
247
+ const indent = match[1].length / 2;
248
+ const level = Math.floor(indent);
249
+ const marker = match[2];
250
+ const isOrdered = !!marker.match(/\d+\./);
251
+ const listType = isOrdered ? 'ordered' : 'unordered';
252
+ if (listCounters[level] === undefined) {
253
+ if (isOrdered) {
254
+ const startNum = parseInt(marker, 10);
255
+ listCounters[level] = isNaN(startNum) ? 0 : startNum - 1;
256
+ }
257
+ else {
258
+ listCounters[level] = 0;
259
+ }
260
+ }
261
+ else {
262
+ listCounters[level]++;
263
+ }
264
+ const children = parseInline(match[3]);
265
+ content.push({
266
+ type: 'list',
267
+ text: children.map(c => c.text || '').join(''),
268
+ metadata: {
269
+ listType,
270
+ indentation: level,
271
+ alignment: alignment || 'left',
272
+ listId,
273
+ itemIndex: listCounters[level]
274
+ },
275
+ children
276
+ });
277
+ }
278
+ }
279
+ continue;
280
+ }
281
+ // Table (Simple Pipe or HTML)
282
+ if ((block.includes('|') && block.match(/\n\s*\|?[-:| ]+\|?\s*\n/)) || block.includes('<table')) {
283
+ if (block.includes('<table')) {
284
+ // Basic HTML table recognition (extracting rows/cells)
285
+ const rows = [];
286
+ const trRegex = /<tr[^>]*>([\s\S]*?)<\/tr>/gi;
287
+ let trMatch;
288
+ while ((trMatch = trRegex.exec(block)) !== null) {
289
+ const tdRegex = /<(?:td|th)([^>]*)>([\s\S]*?)<\/(?:td|th)>/gi;
290
+ let tdMatch;
291
+ const cells = [];
292
+ while ((tdMatch = tdRegex.exec(trMatch[1])) !== null) {
293
+ const attrs = tdMatch[1];
294
+ const contentStr = tdMatch[2].trim();
295
+ const colSpanMatch = attrs.match(/colspan=["']?(\d+)["']?/i);
296
+ const rowSpanMatch = attrs.match(/rowspan=["']?(\d+)["']?/i);
297
+ cells.push({
298
+ type: 'cell',
299
+ metadata: {
300
+ colSpan: colSpanMatch ? parseInt(colSpanMatch[1]) : undefined,
301
+ rowSpan: rowSpanMatch ? parseInt(rowSpanMatch[1]) : undefined
302
+ },
303
+ children: parseInline(contentStr.replace(/<[^>]*>/g, ''))
304
+ });
305
+ }
306
+ if (cells.length > 0)
307
+ rows.push({ type: 'row', children: cells });
308
+ }
309
+ if (rows.length > 0) {
310
+ content.push({ type: 'table', children: rows });
311
+ continue;
312
+ }
313
+ }
314
+ else {
315
+ const lines = block.trim().split('\n');
316
+ const rows = [];
317
+ for (let i = 0; i < lines.length; i++) {
318
+ if (lines[i].match(/^\|?[-:| ]*---[-:| ]*\|?$/))
319
+ continue; // Separator row (requires at least one triple-hyphen)
320
+ const cellsStr = lines[i].replace(/^\||\|$/g, '').split('|');
321
+ const cells = cellsStr.map(c => ({
322
+ type: 'cell',
323
+ children: parseInline(c.trim(), i === 0 ? { bold: true } : {})
324
+ }));
325
+ rows.push({ type: 'row', children: cells });
326
+ }
327
+ content.push({ type: 'table', children: rows });
328
+ continue;
329
+ }
330
+ }
331
+ // Hr
332
+ if (block.match(/^---+|^\*\*\*+|___+$/)) {
333
+ content.push({ type: 'break', metadata: { breakType: 'page' } });
334
+ continue;
335
+ }
336
+ // Paragraph
337
+ content.push({
338
+ type: 'paragraph',
339
+ metadata: { alignment },
340
+ children: parseInline(block.replace(/\n/g, ' '))
341
+ });
342
+ }
343
+ const toTextSync = () => content.map(n => {
344
+ const getText = (node) => {
345
+ if (node.type === 'text' || node.type === 'code')
346
+ return node.text || '';
347
+ if (node.type === 'break')
348
+ return '\n';
349
+ if (node.children) {
350
+ const isBlock = ['table', 'row', 'list', 'sheet', 'slide'].includes(node.type);
351
+ return node.children.map(getText).join(isBlock ? config.newlineDelimiter : '');
352
+ }
353
+ return '';
354
+ };
355
+ return getText(n);
356
+ }).join(config.newlineDelimiter)
357
+ .replace(/\n{3,}/g, '\n\n'); // Normalize excessive whitespace
358
+ return (0, astUtils_js_1.createAST)('md', metadata, content, attachments, config, toTextSync);
359
+ };
360
+ exports.parseMarkdown = parseMarkdown;
@@ -20,7 +20,7 @@
20
20
  *
21
21
  * @module OpenOfficeParser
22
22
  */
23
- import { OfficeParserAST, OfficeParserConfig } from '../types.js';
23
+ import { FullOfficeParserConfig, OfficeParserAST } from '../types.js';
24
24
  /**
25
25
  * Parses an OpenOffice document (.odt, .odp, .ods) and extracts content.
26
26
  *
@@ -28,4 +28,4 @@ import { OfficeParserAST, OfficeParserConfig } from '../types.js';
28
28
  * @param config - Parser configuration
29
29
  * @returns A promise resolving to the parsed AST
30
30
  */
31
- export declare const parseOpenOffice: (buffer: Buffer, config: OfficeParserConfig) => Promise<OfficeParserAST>;
31
+ export declare const parseOpenOffice: (buffer: Buffer, config: FullOfficeParserConfig) => Promise<OfficeParserAST>;