officeparser 6.1.1 → 7.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. package/README.md +301 -26
  2. package/dist/OfficeConverter.d.ts +46 -0
  3. package/dist/OfficeConverter.js +72 -0
  4. package/dist/OfficeGenerator.d.ts +19 -0
  5. package/dist/OfficeGenerator.js +48 -0
  6. package/dist/OfficeParser.d.ts +6 -0
  7. package/dist/OfficeParser.js +74 -31
  8. package/dist/cli.d.ts +3 -1
  9. package/dist/cli.js +106 -22
  10. package/dist/defaults.d.ts +41 -0
  11. package/dist/defaults.js +172 -0
  12. package/dist/generators/BaseGenerator.d.ts +58 -0
  13. package/dist/generators/BaseGenerator.js +107 -0
  14. package/dist/generators/ChunkingGenerator.d.ts +81 -0
  15. package/dist/generators/ChunkingGenerator.js +683 -0
  16. package/dist/generators/CsvGenerator.d.ts +30 -0
  17. package/dist/generators/CsvGenerator.js +233 -0
  18. package/dist/generators/HtmlGenerator.d.ts +37 -0
  19. package/dist/generators/HtmlGenerator.js +1013 -0
  20. package/dist/generators/MarkdownGenerator.d.ts +59 -0
  21. package/dist/generators/MarkdownGenerator.js +481 -0
  22. package/dist/generators/PdfGenerator.d.ts +22 -0
  23. package/dist/generators/PdfGenerator.js +118 -0
  24. package/dist/generators/RtfGenerator.d.ts +15 -0
  25. package/dist/generators/RtfGenerator.js +208 -0
  26. package/dist/generators/TextGenerator.d.ts +13 -0
  27. package/dist/generators/TextGenerator.js +108 -0
  28. package/dist/index.d.ts +11 -3
  29. package/dist/index.js +17 -2
  30. package/dist/index.mjs +2 -2
  31. package/dist/officeparser.browser.d.ts +828 -5
  32. package/dist/officeparser.browser.iife.js +703 -52
  33. package/dist/officeparser.browser.mjs +703 -52
  34. package/dist/parsers/CsvParser.d.ts +9 -0
  35. package/dist/parsers/CsvParser.js +110 -0
  36. package/dist/parsers/ExcelParser.d.ts +2 -2
  37. package/dist/parsers/ExcelParser.js +145 -114
  38. package/dist/parsers/HtmlParser.d.ts +2 -0
  39. package/dist/parsers/HtmlParser.js +539 -0
  40. package/dist/parsers/MarkdownParser.d.ts +2 -0
  41. package/dist/parsers/MarkdownParser.js +360 -0
  42. package/dist/parsers/OpenOfficeParser.d.ts +2 -2
  43. package/dist/parsers/OpenOfficeParser.js +140 -79
  44. package/dist/parsers/PdfParser.d.ts +2 -2
  45. package/dist/parsers/PdfParser.js +52 -49
  46. package/dist/parsers/PowerPointParser.d.ts +2 -2
  47. package/dist/parsers/PowerPointParser.js +20 -23
  48. package/dist/parsers/RtfParser.d.ts +2 -2
  49. package/dist/parsers/RtfParser.js +1291 -1240
  50. package/dist/parsers/WordParser.d.ts +2 -2
  51. package/dist/parsers/WordParser.js +232 -97
  52. package/dist/sbom.cdx.json +99 -99
  53. package/dist/types.d.ts +783 -5
  54. package/dist/types.js +73 -0
  55. package/dist/utils/astUtils.d.ts +16 -0
  56. package/dist/utils/astUtils.js +32 -0
  57. package/dist/utils/configUtils.d.ts +26 -0
  58. package/dist/utils/configUtils.js +140 -0
  59. package/dist/utils/envUtils.d.ts +8 -3
  60. package/dist/utils/envUtils.js +117 -34
  61. package/dist/utils/errorUtils.d.ts +17 -29
  62. package/dist/utils/errorUtils.js +110 -52
  63. package/dist/utils/moduleLoader.js +19 -11
  64. package/dist/utils/ocrUtils.js +2 -1
  65. package/dist/utils/sheetUtils.d.ts +7 -0
  66. package/dist/utils/sheetUtils.js +35 -0
  67. package/dist/utils/styleMapper.d.ts +36 -0
  68. package/dist/utils/styleMapper.js +224 -0
  69. package/dist/utils/xmlUtils.d.ts +0 -8
  70. package/dist/utils/xmlUtils.js +2 -1
  71. package/package.json +26 -7
@@ -0,0 +1,539 @@
1
+ "use strict";
2
+ Object.defineProperty(exports, "__esModule", { value: true });
3
+ exports.parseHtml = void 0;
4
+ const astUtils_js_1 = require("../utils/astUtils.js");
5
+ const parseAttributes = (attrString) => {
6
+ const attrs = {};
7
+ const regex = /([a-zA-Z0-9\-:]+)(?:\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s>]+)))?/g;
8
+ let match;
9
+ while ((match = regex.exec(attrString)) !== null) {
10
+ const name = match[1].toLowerCase();
11
+ const value = match[2] !== undefined ? match[2] : (match[3] !== undefined ? match[3] : (match[4] || ''));
12
+ attrs[name] = value;
13
+ }
14
+ return attrs;
15
+ };
16
+ const parseHtmlTree = (html) => {
17
+ const root = { type: 'element', tagName: 'root', children: [], attributes: {} };
18
+ let current = root;
19
+ let cursor = 0;
20
+ while (cursor < html.length) {
21
+ const tagStart = html.indexOf('<', cursor);
22
+ if (tagStart === -1) {
23
+ const text = html.substring(cursor);
24
+ if (text)
25
+ current.children.push({ type: 'text', text, children: [], parent: current });
26
+ break;
27
+ }
28
+ if (tagStart > cursor) {
29
+ const text = html.substring(cursor, tagStart);
30
+ if (text)
31
+ current.children.push({ type: 'text', text, children: [], parent: current });
32
+ }
33
+ if (html.startsWith('<!--', tagStart)) {
34
+ const commentEnd = html.indexOf('-->', tagStart + 4);
35
+ cursor = commentEnd !== -1 ? commentEnd + 3 : html.length;
36
+ continue;
37
+ }
38
+ const tagEndMatch = html.substring(tagStart).match(/>/);
39
+ if (!tagEndMatch) {
40
+ const text = html.substring(tagStart);
41
+ current.children.push({ type: 'text', text, children: [], parent: current });
42
+ break;
43
+ }
44
+ const tagContent = html.substring(tagStart + 1, tagStart + tagEndMatch.index);
45
+ cursor = tagStart + tagEndMatch.index + 1;
46
+ const isClosing = tagContent.startsWith('/');
47
+ const isSelfClosing = tagContent.endsWith('/');
48
+ const tagCore = tagContent.replace(/^\/|\/$/g, '').trim();
49
+ const firstSpace = tagCore.search(/\s/);
50
+ const tagName = (firstSpace === -1 ? tagCore : tagCore.substring(0, firstSpace)).toLowerCase();
51
+ const attrString = firstSpace === -1 ? '' : tagCore.substring(firstSpace);
52
+ if (!tagName || !tagName.match(/^[a-z0-9\-]+$/)) {
53
+ // Probably not a real tag, e.g., < 5
54
+ current.children.push({ type: 'text', text: `<${tagContent}>`, children: [], parent: current });
55
+ continue;
56
+ }
57
+ if (isClosing) {
58
+ let p = current;
59
+ while (p && p.tagName !== tagName) {
60
+ p = p.parent;
61
+ }
62
+ if (p && p.parent) {
63
+ current = p.parent;
64
+ }
65
+ }
66
+ else {
67
+ const node = {
68
+ type: 'element',
69
+ tagName,
70
+ attributes: parseAttributes(attrString),
71
+ children: [],
72
+ parent: current
73
+ };
74
+ current.children.push(node);
75
+ const voidElements = new Set(['area', 'base', 'br', 'col', 'embed', 'hr', 'img', 'input', 'link', 'meta', 'param', 'source', 'track', 'wbr', '!doctype']);
76
+ if (!isSelfClosing && !voidElements.has(tagName)) {
77
+ current = node;
78
+ if (tagName === 'script' || tagName === 'style') {
79
+ const closeTag = `</${tagName}>`;
80
+ const closeIdx = html.toLowerCase().indexOf(closeTag, cursor);
81
+ if (closeIdx !== -1) {
82
+ node.children.push({
83
+ type: 'text',
84
+ text: html.substring(cursor, closeIdx),
85
+ children: [],
86
+ parent: node
87
+ });
88
+ cursor = closeIdx + closeTag.length;
89
+ current = node.parent;
90
+ }
91
+ }
92
+ }
93
+ }
94
+ }
95
+ return root;
96
+ };
97
+ const parseHtml = async (buffer, config) => {
98
+ const textStr = buffer.toString('utf-8');
99
+ const root = parseHtmlTree(textStr);
100
+ // Find head and body
101
+ let head;
102
+ let body = root;
103
+ const findNode = (node, tag) => {
104
+ if (node.tagName === tag)
105
+ return node;
106
+ for (const child of node.children) {
107
+ const found = findNode(child, tag);
108
+ if (found)
109
+ return found;
110
+ }
111
+ return undefined;
112
+ };
113
+ const htmlNode = findNode(root, 'html');
114
+ if (htmlNode) {
115
+ head = findNode(htmlNode, 'head');
116
+ body = findNode(htmlNode, 'body') || htmlNode;
117
+ }
118
+ const metadata = {};
119
+ const attachments = [];
120
+ if (head) {
121
+ const titleNode = findNode(head, 'title');
122
+ if (titleNode && titleNode.children.length > 0 && titleNode.children[0].text) {
123
+ metadata.title = titleNode.children[0].text;
124
+ }
125
+ const extractMeta = (name) => {
126
+ for (const child of head.children) {
127
+ if (child.tagName === 'meta' && (child.attributes?.name === name || child.attributes?.property === name)) {
128
+ return child.attributes?.content;
129
+ }
130
+ }
131
+ return undefined;
132
+ };
133
+ const author = extractMeta('author');
134
+ if (author)
135
+ metadata.author = author;
136
+ const desc = extractMeta('description');
137
+ if (desc)
138
+ metadata.description = desc;
139
+ const created = extractMeta('dcterms.created');
140
+ if (created)
141
+ metadata.created = new Date(created);
142
+ const modified = extractMeta('dcterms.modified');
143
+ if (modified)
144
+ metadata.modified = new Date(modified);
145
+ const lastMod = extractMeta('lastModifiedBy');
146
+ if (lastMod)
147
+ metadata.lastModifiedBy = lastMod;
148
+ // Custom properties
149
+ const customProps = {};
150
+ for (const child of head.children) {
151
+ if (child.tagName === 'meta' && child.attributes?.name?.startsWith('custom:')) {
152
+ const key = child.attributes.name.substring(7);
153
+ const val = child.attributes.content || '';
154
+ // Try to infer type
155
+ if (val === 'true')
156
+ customProps[key] = true;
157
+ else if (val === 'false')
158
+ customProps[key] = false;
159
+ else if (!isNaN(Number(val)) && val.trim() !== '')
160
+ customProps[key] = Number(val);
161
+ else if (!isNaN(Date.parse(val)) && val.includes(':'))
162
+ customProps[key] = new Date(val);
163
+ else
164
+ customProps[key] = val;
165
+ }
166
+ }
167
+ if (Object.keys(customProps).length > 0)
168
+ metadata.customProperties = customProps;
169
+ }
170
+ const content = [];
171
+ let htmlListIdCounter = 1;
172
+ const parseNode = (node, currentFormatting = {}, listContext) => {
173
+ if (node.type === 'text') {
174
+ let decodedText = (node.text || '')
175
+ .replace(/&nbsp;/g, ' ')
176
+ .replace(/&lt;/g, '<')
177
+ .replace(/&gt;/g, '>')
178
+ .replace(/&amp;/g, '&')
179
+ .replace(/&quot;/g, '"')
180
+ .replace(/&#39;/g, "'");
181
+ if (!config.preserveXmlWhitespace) {
182
+ decodedText = decodedText.replace(/\s+/g, ' ');
183
+ }
184
+ if (!decodedText.trim() && !config.preserveXmlWhitespace)
185
+ return null;
186
+ const textNode = {
187
+ type: 'text',
188
+ text: decodedText,
189
+ formatting: Object.keys(currentFormatting).length > 0 ? { ...currentFormatting } : undefined
190
+ };
191
+ if (config.includeRawContent && node.text) {
192
+ // For text nodes in this manual parser, we just use the decoded text as raw content
193
+ // as we don't have accurate locators for the original source slice
194
+ textNode.rawContent = node.text;
195
+ }
196
+ return textNode;
197
+ }
198
+ if (node.type === 'element' && node.tagName) {
199
+ const tagName = node.tagName;
200
+ const newFormatting = { ...currentFormatting };
201
+ if (tagName === 'b' || tagName === 'strong')
202
+ newFormatting.bold = true;
203
+ if (tagName === 'i' || tagName === 'em')
204
+ newFormatting.italic = true;
205
+ if (tagName === 'u')
206
+ newFormatting.underline = true;
207
+ if (tagName === 'strike' || tagName === 's' || tagName === 'del')
208
+ newFormatting.strikethrough = true;
209
+ if (tagName === 'sub')
210
+ newFormatting.subscript = true;
211
+ if (tagName === 'sup')
212
+ newFormatting.superscript = true;
213
+ if (tagName === 'code')
214
+ newFormatting.font = 'monospace';
215
+ const styleAttr = node.attributes?.style || '';
216
+ const alignAttr = node.attributes?.align || '';
217
+ if (styleAttr || alignAttr) {
218
+ if (styleAttr.includes('font-weight: bold'))
219
+ newFormatting.bold = true;
220
+ if (styleAttr.includes('font-style: italic'))
221
+ newFormatting.italic = true;
222
+ if (styleAttr.includes('text-decoration: underline'))
223
+ newFormatting.underline = true;
224
+ if (styleAttr.includes('text-decoration: line-through'))
225
+ newFormatting.strikethrough = true;
226
+ const colorMatch = styleAttr.match(/color:\s*([^;]+)/);
227
+ if (colorMatch)
228
+ newFormatting.color = colorMatch[1].trim();
229
+ const bgMatch = styleAttr.match(/background-color:\s*([^;]+)/);
230
+ if (bgMatch)
231
+ newFormatting.backgroundColor = bgMatch[1].trim();
232
+ const sizeMatch = styleAttr.match(/font-size:\s*([^;]+)/);
233
+ if (sizeMatch)
234
+ newFormatting.size = sizeMatch[1].trim();
235
+ const fontMatch = styleAttr.match(/font-family:\s*([^;]+)/);
236
+ if (fontMatch)
237
+ newFormatting.font = fontMatch[1].trim().split(',')[0].replace(/['"]/g, '');
238
+ const alignmentMatch = styleAttr.match(/text-align:\s*(left|center|right|justify)/);
239
+ if (alignmentMatch) {
240
+ newFormatting.alignment = alignmentMatch[1].toLowerCase();
241
+ }
242
+ else if (alignAttr) {
243
+ const align = alignAttr.toLowerCase();
244
+ if (['left', 'center', 'right', 'justify'].includes(align)) {
245
+ newFormatting.alignment = align;
246
+ }
247
+ }
248
+ }
249
+ const anchorIds = node.attributes?.id ? [node.attributes.id] : [];
250
+ const parseChildren = (n, fmt, lCtx) => {
251
+ const kids = [];
252
+ for (const child of n.children) {
253
+ const parsed = parseNode(child, fmt, lCtx);
254
+ if (parsed) {
255
+ if (Array.isArray(parsed))
256
+ kids.push(...parsed);
257
+ else
258
+ kids.push(parsed);
259
+ }
260
+ }
261
+ return kids;
262
+ };
263
+ // Skip structural containers produced by HtmlGenerator to avoid deep AST nesting
264
+ if (tagName === 'div' && (node.attributes?.class === 'container' ||
265
+ node.attributes?.class === 'spreadsheet-container' ||
266
+ node.attributes?.class === 'presentation-container' ||
267
+ node.attributes?.class === 'pdf-container' ||
268
+ node.attributes?.class === 'metadata-summary' ||
269
+ node.attributes?.class === 'image-container' ||
270
+ node.attributes?.class === 'chart-container' ||
271
+ node.attributes?.class === 'table-container' ||
272
+ node.attributes?.class === 'caption' ||
273
+ node.attributes?.class === 'sheet' ||
274
+ node.attributes?.class === 'page' ||
275
+ node.attributes?.class === 'slide' ||
276
+ node.attributes?.class === 'note-content')) {
277
+ return parseChildren(node, newFormatting, listContext);
278
+ }
279
+ if (tagName === 'article') {
280
+ return parseChildren(node, newFormatting, listContext);
281
+ }
282
+ if (tagName === 'p' || tagName === 'div') {
283
+ const children = parseChildren(node, newFormatting, listContext);
284
+ // If it's a div and contains block elements, return children directly
285
+ const hasBlockElements = children.some(c => ['paragraph', 'table', 'heading', 'list', 'image', 'chart', 'code'].includes(c.type));
286
+ if (tagName === 'div' && hasBlockElements) {
287
+ return children;
288
+ }
289
+ // Flatten nested paragraphs to avoid deep AST nesting (e.g. from notes)
290
+ const flattenedChildren = [];
291
+ for (const child of children) {
292
+ if (child.type === 'paragraph' && child.children) {
293
+ flattenedChildren.push(...child.children);
294
+ }
295
+ else {
296
+ flattenedChildren.push(child);
297
+ }
298
+ }
299
+ const pNode = {
300
+ type: 'paragraph',
301
+ metadata: { alignment: newFormatting.alignment, anchorIds: anchorIds.length > 0 ? anchorIds : undefined },
302
+ children: flattenedChildren
303
+ };
304
+ if (config.includeRawContent) {
305
+ // Note: Since this is a manual parser without locators, we can't easily get the original source slice.
306
+ // We'll skip rawContent for structural nodes here unless we want to implement index tracking in parseHtmlTree.
307
+ }
308
+ return pNode;
309
+ }
310
+ if (tagName.match(/^h[1-6]$/)) {
311
+ const level = parseInt(tagName.substring(1));
312
+ const hNode = {
313
+ type: 'heading',
314
+ metadata: { level, alignment: newFormatting.alignment, anchorIds: anchorIds.length > 0 ? anchorIds : undefined },
315
+ children: parseChildren(node, newFormatting, listContext)
316
+ };
317
+ return hNode;
318
+ }
319
+ if (tagName === 'ul' || tagName === 'ol') {
320
+ const isNewTopLevel = !listContext;
321
+ const newListContext = {
322
+ listId: isNewTopLevel ? `html-list-${htmlListIdCounter++}` : listContext.listId,
323
+ type: tagName === 'ol' ? 'ordered' : 'unordered',
324
+ level: isNewTopLevel ? 0 : listContext.level + 1,
325
+ counters: isNewTopLevel ? {} : { ...listContext.counters } // Clone to avoid side effects on parent levels
326
+ };
327
+ // Initialize counter for this level
328
+ if (tagName === 'ol' && node.attributes?.start) {
329
+ const start = parseInt(node.attributes.start, 10);
330
+ newListContext.counters[newListContext.level] = isNaN(start) ? 0 : start - 1;
331
+ }
332
+ else {
333
+ newListContext.counters[newListContext.level] = 0;
334
+ }
335
+ return parseChildren(node, currentFormatting, newListContext);
336
+ }
337
+ if (tagName === 'li') {
338
+ if (listContext) {
339
+ if (node.attributes?.value) {
340
+ const val = parseInt(node.attributes.value, 10);
341
+ if (!isNaN(val))
342
+ listContext.counters[listContext.level] = val;
343
+ }
344
+ else {
345
+ listContext.counters[listContext.level]++;
346
+ }
347
+ }
348
+ const children = parseChildren(node, newFormatting, listContext);
349
+ const nestedLists = children.filter(c => c.type === 'list');
350
+ const selfChildren = children.filter(c => c.type !== 'list');
351
+ const selfNode = {
352
+ type: 'list',
353
+ text: selfChildren.map(c => c.text || '').join(''),
354
+ metadata: {
355
+ listType: listContext?.type || 'unordered',
356
+ indentation: listContext?.level || 0,
357
+ alignment: newFormatting.alignment || 'left',
358
+ listId: listContext?.listId || 'html-list-none',
359
+ itemIndex: (listContext?.counters[listContext.level] ?? 1) - 1,
360
+ anchorIds: anchorIds.length > 0 ? anchorIds : undefined
361
+ },
362
+ children: selfChildren
363
+ };
364
+ return [selfNode, ...nestedLists];
365
+ }
366
+ if (tagName === 'table') {
367
+ const tableNode = {
368
+ type: 'table',
369
+ metadata: { anchorIds: anchorIds.length > 0 ? anchorIds : undefined },
370
+ children: parseChildren(node, newFormatting, listContext)
371
+ };
372
+ if (config.includeRawContent) {
373
+ tableNode.rawContent = '<table>...</table>';
374
+ }
375
+ return tableNode;
376
+ }
377
+ if (tagName === 'tr') {
378
+ const rowNode = {
379
+ type: 'row',
380
+ children: parseChildren(node, newFormatting, listContext)
381
+ };
382
+ if (config.includeRawContent) {
383
+ rowNode.rawContent = '<tr>...</tr>';
384
+ }
385
+ return rowNode;
386
+ }
387
+ if (tagName === 'td' || tagName === 'th') {
388
+ const cellNode = {
389
+ type: 'cell',
390
+ children: parseChildren(node, newFormatting, listContext)
391
+ };
392
+ if (config.includeRawContent) {
393
+ cellNode.rawContent = '<td>...</td>';
394
+ }
395
+ return cellNode;
396
+ }
397
+ if (tagName === 'img') {
398
+ const src = node.attributes?.src;
399
+ const alt = node.attributes?.alt;
400
+ let imageNode;
401
+ if (src?.startsWith('data:')) {
402
+ const match = src.match(/^data:([^;]+);base64,(.*)$/);
403
+ if (match && config.extractAttachments) {
404
+ const mimeType = match[1];
405
+ const data = match[2];
406
+ const name = `image_${attachments.length + 1}.${mimeType.split('/')[1]}`;
407
+ attachments.push({
408
+ type: 'image',
409
+ mimeType,
410
+ data,
411
+ name,
412
+ extension: mimeType.split('/')[1]
413
+ });
414
+ imageNode = {
415
+ type: 'image',
416
+ metadata: {
417
+ attachmentName: name,
418
+ altText: alt
419
+ }
420
+ };
421
+ }
422
+ else {
423
+ imageNode = {
424
+ type: 'image',
425
+ metadata: {
426
+ url: src,
427
+ altText: alt
428
+ }
429
+ };
430
+ }
431
+ }
432
+ else {
433
+ imageNode = {
434
+ type: 'image',
435
+ metadata: {
436
+ url: src,
437
+ altText: alt,
438
+ anchorIds: anchorIds.length > 0 ? anchorIds : undefined
439
+ }
440
+ };
441
+ }
442
+ if (config.includeRawContent) {
443
+ imageNode.rawContent = '<img>';
444
+ }
445
+ return imageNode;
446
+ }
447
+ if (tagName === 'a') {
448
+ const href = node.attributes?.href;
449
+ const children = parseChildren(node, newFormatting, listContext);
450
+ if (href) {
451
+ const linkType = href.startsWith('#') ? 'internal' : 'external';
452
+ children.forEach(c => {
453
+ if (c.type === 'text') {
454
+ c.metadata = { ...c.metadata, link: href, linkType };
455
+ }
456
+ });
457
+ }
458
+ return children;
459
+ }
460
+ if (tagName === 'br') {
461
+ const brNode = { type: 'break', metadata: { breakType: 'textWrapping' } };
462
+ if (config.includeRawContent) {
463
+ brNode.rawContent = '<br/>';
464
+ }
465
+ return brNode;
466
+ }
467
+ if (tagName === 'pre') {
468
+ const codeNode = node.children.find(c => c.tagName === 'code');
469
+ let language;
470
+ let codeText = '';
471
+ if (codeNode) {
472
+ const classAttr = codeNode.attributes?.class || '';
473
+ const langMatch = classAttr.split(' ').find((c) => c.startsWith('language-'));
474
+ if (langMatch)
475
+ language = langMatch.replace('language-', '');
476
+ codeText = codeNode.children.map(c => c.text || '').join('');
477
+ }
478
+ else {
479
+ codeText = node.children.map(c => c.text || '').join('');
480
+ }
481
+ const preNode = {
482
+ type: 'code',
483
+ text: codeText,
484
+ metadata: { language, anchorIds: anchorIds.length > 0 ? anchorIds : undefined }
485
+ };
486
+ if (config.includeRawContent) {
487
+ preNode.rawContent = '<pre>...</pre>';
488
+ }
489
+ return preNode;
490
+ }
491
+ if (tagName === 'script' || tagName === 'style' || tagName === '!doctype') {
492
+ return null;
493
+ }
494
+ return parseChildren(node, newFormatting, listContext);
495
+ }
496
+ return null;
497
+ };
498
+ for (const child of body.children) {
499
+ const parsed = parseNode(child);
500
+ if (parsed) {
501
+ if (Array.isArray(parsed)) {
502
+ parsed.forEach(p => {
503
+ if (p.type === 'text') {
504
+ // Wrap direct body text in paragraphs
505
+ content.push({ type: 'paragraph', children: [p] });
506
+ }
507
+ else {
508
+ content.push(p);
509
+ }
510
+ });
511
+ }
512
+ else {
513
+ if (parsed.type === 'text') {
514
+ content.push({ type: 'paragraph', children: [parsed] });
515
+ }
516
+ else {
517
+ content.push(parsed);
518
+ }
519
+ }
520
+ }
521
+ }
522
+ const toTextSync = () => content.map(n => {
523
+ const getText = (node) => {
524
+ if (node.type === 'text' || node.type === 'code')
525
+ return node.text || '';
526
+ if (node.type === 'break')
527
+ return '\n';
528
+ if (node.children) {
529
+ const isBlock = ['table', 'row', 'list', 'sheet', 'slide'].includes(node.type);
530
+ return node.children.map(getText).join(isBlock ? config.newlineDelimiter : '');
531
+ }
532
+ return '';
533
+ };
534
+ return getText(n);
535
+ }).join(config.newlineDelimiter)
536
+ .replace(/\n{3,}/g, '\n\n'); // Normalize excessive whitespace
537
+ return (0, astUtils_js_1.createAST)('html', metadata, content, attachments, config, toTextSync);
538
+ };
539
+ exports.parseHtml = parseHtml;
@@ -0,0 +1,2 @@
1
+ import { FullOfficeParserConfig, OfficeParserAST } from '../types.js';
2
+ export declare const parseMarkdown: (buffer: Buffer, config: FullOfficeParserConfig) => Promise<OfficeParserAST>;