officeparser 6.1.0 → 7.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +284 -86
- package/dist/OfficeConverter.d.ts +46 -0
- package/dist/OfficeConverter.js +72 -0
- package/dist/OfficeGenerator.d.ts +19 -0
- package/dist/OfficeGenerator.js +48 -0
- package/dist/OfficeParser.d.ts +6 -0
- package/dist/OfficeParser.js +55 -28
- package/dist/cli.d.ts +3 -1
- package/dist/cli.js +107 -22
- package/dist/defaults.d.ts +41 -0
- package/dist/defaults.js +172 -0
- package/dist/generators/BaseGenerator.d.ts +58 -0
- package/dist/generators/BaseGenerator.js +107 -0
- package/dist/generators/ChunkingGenerator.d.ts +81 -0
- package/dist/generators/ChunkingGenerator.js +683 -0
- package/dist/generators/CsvGenerator.d.ts +30 -0
- package/dist/generators/CsvGenerator.js +233 -0
- package/dist/generators/HtmlGenerator.d.ts +37 -0
- package/dist/generators/HtmlGenerator.js +1013 -0
- package/dist/generators/MarkdownGenerator.d.ts +59 -0
- package/dist/generators/MarkdownGenerator.js +481 -0
- package/dist/generators/PdfGenerator.d.ts +22 -0
- package/dist/generators/PdfGenerator.js +118 -0
- package/dist/generators/RtfGenerator.d.ts +15 -0
- package/dist/generators/RtfGenerator.js +208 -0
- package/dist/generators/TextGenerator.d.ts +13 -0
- package/dist/generators/TextGenerator.js +108 -0
- package/dist/index.d.ts +11 -3
- package/dist/index.js +17 -2
- package/dist/index.mjs +2 -2
- package/dist/officeparser.browser.d.ts +878 -5
- package/dist/officeparser.browser.iife.js +703 -49
- package/dist/officeparser.browser.mjs +703 -49
- package/dist/parsers/CsvParser.d.ts +9 -0
- package/dist/parsers/CsvParser.js +110 -0
- package/dist/parsers/ExcelParser.d.ts +2 -2
- package/dist/parsers/ExcelParser.js +145 -114
- package/dist/parsers/HtmlParser.d.ts +2 -0
- package/dist/parsers/HtmlParser.js +539 -0
- package/dist/parsers/MarkdownParser.d.ts +2 -0
- package/dist/parsers/MarkdownParser.js +360 -0
- package/dist/parsers/OpenOfficeParser.d.ts +2 -2
- package/dist/parsers/OpenOfficeParser.js +237 -128
- package/dist/parsers/PdfParser.d.ts +2 -2
- package/dist/parsers/PdfParser.js +52 -49
- package/dist/parsers/PowerPointParser.d.ts +2 -2
- package/dist/parsers/PowerPointParser.js +132 -123
- package/dist/parsers/RtfParser.d.ts +22 -2
- package/dist/parsers/RtfParser.js +1398 -1282
- package/dist/parsers/WordParser.d.ts +3 -2
- package/dist/parsers/WordParser.js +333 -115
- package/dist/sbom.cdx.json +103 -103
- package/dist/types.d.ts +833 -5
- package/dist/types.js +71 -0
- package/dist/utils/astUtils.d.ts +16 -0
- package/dist/utils/astUtils.js +32 -0
- package/dist/utils/configUtils.d.ts +26 -0
- package/dist/utils/configUtils.js +140 -0
- package/dist/utils/envUtils.js +56 -2
- package/dist/utils/errorUtils.d.ts +17 -29
- package/dist/utils/errorUtils.js +109 -52
- package/dist/utils/moduleLoader.js +15 -9
- package/dist/utils/ocrUtils.js +2 -1
- package/dist/utils/sheetUtils.d.ts +7 -0
- package/dist/utils/sheetUtils.js +35 -0
- package/dist/utils/styleMapper.d.ts +36 -0
- package/dist/utils/styleMapper.js +224 -0
- package/dist/utils/xmlUtils.d.ts +0 -8
- package/dist/utils/xmlUtils.js +2 -1
- package/package.json +28 -9
|
@@ -0,0 +1,360 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.parseMarkdown = void 0;
|
|
4
|
+
const astUtils_js_1 = require("../utils/astUtils.js");
|
|
5
|
+
const parseMarkdown = async (buffer, config) => {
|
|
6
|
+
let textStr = buffer.toString('utf-8');
|
|
7
|
+
textStr = textStr.replace(/\r\n/g, '\n');
|
|
8
|
+
const content = [];
|
|
9
|
+
const metadata = {};
|
|
10
|
+
const attachments = [];
|
|
11
|
+
// Parse YAML Front Matter
|
|
12
|
+
if (textStr.startsWith('---\n')) {
|
|
13
|
+
const endIdx = textStr.indexOf('\n---\n', 4);
|
|
14
|
+
if (endIdx !== -1) {
|
|
15
|
+
const frontMatter = textStr.substring(4, endIdx);
|
|
16
|
+
textStr = textStr.substring(endIdx + 5);
|
|
17
|
+
const lines = frontMatter.split('\n');
|
|
18
|
+
const customProps = {};
|
|
19
|
+
for (const line of lines) {
|
|
20
|
+
const match = line.match(/^([^:]+):\s*(.*)$/);
|
|
21
|
+
if (match) {
|
|
22
|
+
const key = match[1].trim();
|
|
23
|
+
let val = match[2].trim().replace(/^"(.*)"$/, '$1');
|
|
24
|
+
if (key === 'title')
|
|
25
|
+
metadata.title = val;
|
|
26
|
+
else if (key === 'author')
|
|
27
|
+
metadata.author = val;
|
|
28
|
+
else if (key === 'created')
|
|
29
|
+
metadata.created = new Date(val);
|
|
30
|
+
else if (key === 'modified')
|
|
31
|
+
metadata.modified = new Date(val);
|
|
32
|
+
else if (key === 'description')
|
|
33
|
+
metadata.description = val;
|
|
34
|
+
else {
|
|
35
|
+
// Try to infer type for custom props
|
|
36
|
+
if (val === 'true')
|
|
37
|
+
customProps[key] = true;
|
|
38
|
+
else if (val === 'false')
|
|
39
|
+
customProps[key] = false;
|
|
40
|
+
else if (!isNaN(Number(val)) && val !== '')
|
|
41
|
+
customProps[key] = Number(val);
|
|
42
|
+
else
|
|
43
|
+
customProps[key] = val;
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
}
|
|
47
|
+
if (Object.keys(customProps).length > 0)
|
|
48
|
+
metadata.customProperties = customProps;
|
|
49
|
+
}
|
|
50
|
+
}
|
|
51
|
+
// Extract code blocks first to protect their contents
|
|
52
|
+
const codeBlocks = [];
|
|
53
|
+
textStr = textStr.replace(/^```(\w*)\n([\s\S]*?)\n```/gm, (match, lang, code) => {
|
|
54
|
+
const id = `__CODE_BLOCK_${codeBlocks.length}__`;
|
|
55
|
+
codeBlocks.push(JSON.stringify({ lang, code }));
|
|
56
|
+
return `\n\n${id}\n\n`;
|
|
57
|
+
});
|
|
58
|
+
const parseInline = (text, currentFormatting = {}) => {
|
|
59
|
+
const nodes = [];
|
|
60
|
+
// Regex matches: 1=!, 2=alt, 3=url | 4=bold | 5=italic | 6=strike | 7=code | 8=underline | 9=subscript | 10=superscript
|
|
61
|
+
const regex = /(!?)\[(.*?)\]\((.*?)\)|\*\*(.+?)\*\*|\*(.+?)\*|~~(.+?)~~|`(.+?)`|<u>(.+?)<\/u>|<sub>(.+?)<\/sub>|<sup>(.+?)<\/sup>/g;
|
|
62
|
+
let lastIndex = 0;
|
|
63
|
+
let match;
|
|
64
|
+
while ((match = regex.exec(text)) !== null) {
|
|
65
|
+
if (match.index > lastIndex) {
|
|
66
|
+
nodes.push({ type: 'text', text: text.substring(lastIndex, match.index), formatting: Object.keys(currentFormatting).length > 0 ? { ...currentFormatting } : undefined });
|
|
67
|
+
}
|
|
68
|
+
if (match[2] !== undefined) { // Image or Link
|
|
69
|
+
const isImage = match[1] === '!';
|
|
70
|
+
const altText = match[2];
|
|
71
|
+
const url = match[3];
|
|
72
|
+
if (isImage) {
|
|
73
|
+
if (url.startsWith('data:')) {
|
|
74
|
+
const dataMatch = url.match(/^data:([^;]+);base64,(.*)$/);
|
|
75
|
+
if (dataMatch && config.extractAttachments) {
|
|
76
|
+
const mimeType = dataMatch[1];
|
|
77
|
+
const data = dataMatch[2];
|
|
78
|
+
const name = `image_${attachments.length + 1}.${mimeType.split('/')[1]}`;
|
|
79
|
+
attachments.push({
|
|
80
|
+
type: 'image',
|
|
81
|
+
mimeType,
|
|
82
|
+
data,
|
|
83
|
+
name,
|
|
84
|
+
extension: mimeType.split('/')[1]
|
|
85
|
+
});
|
|
86
|
+
nodes.push({ type: 'image', metadata: { attachmentName: name, altText } });
|
|
87
|
+
}
|
|
88
|
+
else {
|
|
89
|
+
nodes.push({ type: 'image', metadata: { url, altText } });
|
|
90
|
+
}
|
|
91
|
+
}
|
|
92
|
+
else {
|
|
93
|
+
nodes.push({ type: 'image', metadata: { url, altText } });
|
|
94
|
+
}
|
|
95
|
+
}
|
|
96
|
+
else {
|
|
97
|
+
const linkNodes = parseInline(altText, currentFormatting);
|
|
98
|
+
linkNodes.forEach(n => {
|
|
99
|
+
if (n.type === 'text') {
|
|
100
|
+
n.metadata = { link: url, linkType: 'external' };
|
|
101
|
+
}
|
|
102
|
+
});
|
|
103
|
+
nodes.push(...linkNodes);
|
|
104
|
+
}
|
|
105
|
+
}
|
|
106
|
+
else if (match[4]) { // Bold
|
|
107
|
+
nodes.push(...parseInline(match[4], { ...currentFormatting, bold: true }));
|
|
108
|
+
}
|
|
109
|
+
else if (match[5]) { // Italic
|
|
110
|
+
nodes.push(...parseInline(match[5], { ...currentFormatting, italic: true }));
|
|
111
|
+
}
|
|
112
|
+
else if (match[6]) { // Strikethrough
|
|
113
|
+
nodes.push(...parseInline(match[6], { ...currentFormatting, strikethrough: true }));
|
|
114
|
+
}
|
|
115
|
+
else if (match[7]) { // Inline Code
|
|
116
|
+
nodes.push({ type: 'text', text: match[7], formatting: { ...currentFormatting, font: 'monospace' } });
|
|
117
|
+
}
|
|
118
|
+
else if (match[8]) { // Underline
|
|
119
|
+
nodes.push(...parseInline(match[8], { ...currentFormatting, underline: true }));
|
|
120
|
+
}
|
|
121
|
+
else if (match[9]) { // Subscript
|
|
122
|
+
nodes.push(...parseInline(match[9], { ...currentFormatting, subscript: true }));
|
|
123
|
+
}
|
|
124
|
+
else if (match[10]) { // Superscript
|
|
125
|
+
nodes.push(...parseInline(match[10], { ...currentFormatting, superscript: true }));
|
|
126
|
+
}
|
|
127
|
+
lastIndex = regex.lastIndex;
|
|
128
|
+
}
|
|
129
|
+
if (lastIndex < text.length) {
|
|
130
|
+
nodes.push({ type: 'text', text: text.substring(lastIndex), formatting: Object.keys(currentFormatting).length > 0 ? { ...currentFormatting } : undefined });
|
|
131
|
+
}
|
|
132
|
+
return nodes;
|
|
133
|
+
};
|
|
134
|
+
const rawBlocks = textStr.split(/\n\n+/);
|
|
135
|
+
const blocks = [];
|
|
136
|
+
// Sub-split blocks that contain headings or lists without double newlines
|
|
137
|
+
for (const rawBlock of rawBlocks) {
|
|
138
|
+
if (!rawBlock.trim())
|
|
139
|
+
continue;
|
|
140
|
+
// Match headings or lists that might be joined with other text via single newline
|
|
141
|
+
const lines = rawBlock.split('\n');
|
|
142
|
+
let currentSubBlock = [];
|
|
143
|
+
for (const line of lines) {
|
|
144
|
+
const isHeading = !!line.match(/^(?:<a[^>]*><\/a>)*\s*#{1,6}\s+/);
|
|
145
|
+
const isList = !!line.match(/^(\s*)([-*+]|\d+\.)\s+/);
|
|
146
|
+
const isHtmlTag = !!line.match(/^<\/?div[^>]*>$/i);
|
|
147
|
+
const prevWasList = currentSubBlock.length > 0 && !!currentSubBlock[currentSubBlock.length - 1].match(/^(\s*)([-*+]|\d+\.)\s+/);
|
|
148
|
+
// Split if:
|
|
149
|
+
// 1. Current line is a heading
|
|
150
|
+
// 2. Current line is a list item but previous was NOT
|
|
151
|
+
// 3. Current line is NOT a list item but previous WAS
|
|
152
|
+
// 4. Current line is an HTML tag (div)
|
|
153
|
+
if ((isHeading || isHtmlTag || (isList !== prevWasList)) && currentSubBlock.length > 0) {
|
|
154
|
+
blocks.push(currentSubBlock.join('\n'));
|
|
155
|
+
currentSubBlock = [];
|
|
156
|
+
}
|
|
157
|
+
currentSubBlock.push(line);
|
|
158
|
+
// Headings and HTML tags are single-line blocks for our state machine
|
|
159
|
+
if (isHeading || isHtmlTag) {
|
|
160
|
+
blocks.push(currentSubBlock.join('\n'));
|
|
161
|
+
currentSubBlock = [];
|
|
162
|
+
}
|
|
163
|
+
}
|
|
164
|
+
if (currentSubBlock.length > 0) {
|
|
165
|
+
blocks.push(currentSubBlock.join('\n'));
|
|
166
|
+
}
|
|
167
|
+
}
|
|
168
|
+
let listIdCounter = 1;
|
|
169
|
+
let currentAlignment = undefined;
|
|
170
|
+
for (let block of blocks) {
|
|
171
|
+
block = block.trim();
|
|
172
|
+
if (!block)
|
|
173
|
+
continue;
|
|
174
|
+
// Check for alignment wrapper start/end
|
|
175
|
+
const alignStartMatch = block.match(/^<div\s+(?:style="text-align:\s*(left|center|right|justify);?"|align="(left|center|right|justify)")>$/i);
|
|
176
|
+
if (alignStartMatch) {
|
|
177
|
+
currentAlignment = (alignStartMatch[1] || alignStartMatch[2]).toLowerCase();
|
|
178
|
+
continue;
|
|
179
|
+
}
|
|
180
|
+
if (block.match(/^<\/div>$/i)) {
|
|
181
|
+
currentAlignment = undefined;
|
|
182
|
+
continue;
|
|
183
|
+
}
|
|
184
|
+
let alignment = currentAlignment;
|
|
185
|
+
// Check for single-line alignment wrapper (for compatibility)
|
|
186
|
+
const alignMatch = block.match(/^<div\s+(?:style="text-align:\s*(left|center|right|justify);?"|align="(left|center|right|justify)")>\s*([\s\S]*?)\s*<\/div>$/i);
|
|
187
|
+
if (alignMatch) {
|
|
188
|
+
alignment = (alignMatch[1] || alignMatch[2]).toLowerCase();
|
|
189
|
+
block = alignMatch[3];
|
|
190
|
+
}
|
|
191
|
+
// Code Block
|
|
192
|
+
const codeMatch = block.match(/^__CODE_BLOCK_(\d+)__$/);
|
|
193
|
+
if (codeMatch) {
|
|
194
|
+
const data = JSON.parse(codeBlocks[parseInt(codeMatch[1])]);
|
|
195
|
+
content.push({
|
|
196
|
+
type: 'code',
|
|
197
|
+
text: data.code,
|
|
198
|
+
metadata: { language: data.lang }
|
|
199
|
+
});
|
|
200
|
+
continue;
|
|
201
|
+
}
|
|
202
|
+
// Heading (allowing for leading HTML anchors and trailing {#anchor})
|
|
203
|
+
const headingMatch = block.match(/^((?:<a[^>]*><\/a>)*)\s*(#{1,6})\s+(.*?)(?:\s+\{#([^}]+)\})?\s*$/s);
|
|
204
|
+
if (headingMatch) {
|
|
205
|
+
const leadingAnchorsRaw = headingMatch[1];
|
|
206
|
+
const rawText = headingMatch[3];
|
|
207
|
+
const explicitAnchor = headingMatch[4];
|
|
208
|
+
const anchorIds = [];
|
|
209
|
+
if (leadingAnchorsRaw) {
|
|
210
|
+
const idMatches = leadingAnchorsRaw.matchAll(/<a\s+name="([^"]+)"/gi);
|
|
211
|
+
for (const m of idMatches)
|
|
212
|
+
anchorIds.push(m[1]);
|
|
213
|
+
}
|
|
214
|
+
if (explicitAnchor)
|
|
215
|
+
anchorIds.push(explicitAnchor);
|
|
216
|
+
const children = parseInline(rawText);
|
|
217
|
+
content.push({
|
|
218
|
+
type: 'heading',
|
|
219
|
+
text: children.map(c => c.text || '').join(''),
|
|
220
|
+
metadata: {
|
|
221
|
+
level: headingMatch[2].length,
|
|
222
|
+
alignment,
|
|
223
|
+
anchorIds: anchorIds.length > 0 ? anchorIds : undefined
|
|
224
|
+
},
|
|
225
|
+
children
|
|
226
|
+
});
|
|
227
|
+
continue;
|
|
228
|
+
}
|
|
229
|
+
// Blockquote
|
|
230
|
+
const quoteMatch = block.match(/^>\s+(.*)$/s);
|
|
231
|
+
if (quoteMatch) {
|
|
232
|
+
content.push({
|
|
233
|
+
type: 'paragraph',
|
|
234
|
+
metadata: { style: 'Quote' },
|
|
235
|
+
children: parseInline(quoteMatch[1].replace(/^>\s+/gm, ''))
|
|
236
|
+
});
|
|
237
|
+
continue;
|
|
238
|
+
}
|
|
239
|
+
// Lists
|
|
240
|
+
if (block.match(/^(\s*)([-*+]|\d+\.)\s+/)) {
|
|
241
|
+
const lines = block.split('\n');
|
|
242
|
+
const listId = `md-list-${listIdCounter++}`;
|
|
243
|
+
const listCounters = {};
|
|
244
|
+
for (const line of lines) {
|
|
245
|
+
const match = line.match(/^(\s*)([-*+]|\d+\.)\s+(.*)$/);
|
|
246
|
+
if (match) {
|
|
247
|
+
const indent = match[1].length / 2;
|
|
248
|
+
const level = Math.floor(indent);
|
|
249
|
+
const marker = match[2];
|
|
250
|
+
const isOrdered = !!marker.match(/\d+\./);
|
|
251
|
+
const listType = isOrdered ? 'ordered' : 'unordered';
|
|
252
|
+
if (listCounters[level] === undefined) {
|
|
253
|
+
if (isOrdered) {
|
|
254
|
+
const startNum = parseInt(marker, 10);
|
|
255
|
+
listCounters[level] = isNaN(startNum) ? 0 : startNum - 1;
|
|
256
|
+
}
|
|
257
|
+
else {
|
|
258
|
+
listCounters[level] = 0;
|
|
259
|
+
}
|
|
260
|
+
}
|
|
261
|
+
else {
|
|
262
|
+
listCounters[level]++;
|
|
263
|
+
}
|
|
264
|
+
const children = parseInline(match[3]);
|
|
265
|
+
content.push({
|
|
266
|
+
type: 'list',
|
|
267
|
+
text: children.map(c => c.text || '').join(''),
|
|
268
|
+
metadata: {
|
|
269
|
+
listType,
|
|
270
|
+
indentation: level,
|
|
271
|
+
alignment: alignment || 'left',
|
|
272
|
+
listId,
|
|
273
|
+
itemIndex: listCounters[level]
|
|
274
|
+
},
|
|
275
|
+
children
|
|
276
|
+
});
|
|
277
|
+
}
|
|
278
|
+
}
|
|
279
|
+
continue;
|
|
280
|
+
}
|
|
281
|
+
// Table (Simple Pipe or HTML)
|
|
282
|
+
if ((block.includes('|') && block.match(/\n\s*\|?[-:| ]+\|?\s*\n/)) || block.includes('<table')) {
|
|
283
|
+
if (block.includes('<table')) {
|
|
284
|
+
// Basic HTML table recognition (extracting rows/cells)
|
|
285
|
+
const rows = [];
|
|
286
|
+
const trRegex = /<tr[^>]*>([\s\S]*?)<\/tr>/gi;
|
|
287
|
+
let trMatch;
|
|
288
|
+
while ((trMatch = trRegex.exec(block)) !== null) {
|
|
289
|
+
const tdRegex = /<(?:td|th)([^>]*)>([\s\S]*?)<\/(?:td|th)>/gi;
|
|
290
|
+
let tdMatch;
|
|
291
|
+
const cells = [];
|
|
292
|
+
while ((tdMatch = tdRegex.exec(trMatch[1])) !== null) {
|
|
293
|
+
const attrs = tdMatch[1];
|
|
294
|
+
const contentStr = tdMatch[2].trim();
|
|
295
|
+
const colSpanMatch = attrs.match(/colspan=["']?(\d+)["']?/i);
|
|
296
|
+
const rowSpanMatch = attrs.match(/rowspan=["']?(\d+)["']?/i);
|
|
297
|
+
cells.push({
|
|
298
|
+
type: 'cell',
|
|
299
|
+
metadata: {
|
|
300
|
+
colSpan: colSpanMatch ? parseInt(colSpanMatch[1]) : undefined,
|
|
301
|
+
rowSpan: rowSpanMatch ? parseInt(rowSpanMatch[1]) : undefined
|
|
302
|
+
},
|
|
303
|
+
children: parseInline(contentStr.replace(/<[^>]*>/g, ''))
|
|
304
|
+
});
|
|
305
|
+
}
|
|
306
|
+
if (cells.length > 0)
|
|
307
|
+
rows.push({ type: 'row', children: cells });
|
|
308
|
+
}
|
|
309
|
+
if (rows.length > 0) {
|
|
310
|
+
content.push({ type: 'table', children: rows });
|
|
311
|
+
continue;
|
|
312
|
+
}
|
|
313
|
+
}
|
|
314
|
+
else {
|
|
315
|
+
const lines = block.trim().split('\n');
|
|
316
|
+
const rows = [];
|
|
317
|
+
for (let i = 0; i < lines.length; i++) {
|
|
318
|
+
if (lines[i].match(/^\|?[-:| ]*---[-:| ]*\|?$/))
|
|
319
|
+
continue; // Separator row (requires at least one triple-hyphen)
|
|
320
|
+
const cellsStr = lines[i].replace(/^\||\|$/g, '').split('|');
|
|
321
|
+
const cells = cellsStr.map(c => ({
|
|
322
|
+
type: 'cell',
|
|
323
|
+
children: parseInline(c.trim(), i === 0 ? { bold: true } : {})
|
|
324
|
+
}));
|
|
325
|
+
rows.push({ type: 'row', children: cells });
|
|
326
|
+
}
|
|
327
|
+
content.push({ type: 'table', children: rows });
|
|
328
|
+
continue;
|
|
329
|
+
}
|
|
330
|
+
}
|
|
331
|
+
// Hr
|
|
332
|
+
if (block.match(/^---+|^\*\*\*+|___+$/)) {
|
|
333
|
+
content.push({ type: 'break', metadata: { breakType: 'page' } });
|
|
334
|
+
continue;
|
|
335
|
+
}
|
|
336
|
+
// Paragraph
|
|
337
|
+
content.push({
|
|
338
|
+
type: 'paragraph',
|
|
339
|
+
metadata: { alignment },
|
|
340
|
+
children: parseInline(block.replace(/\n/g, ' '))
|
|
341
|
+
});
|
|
342
|
+
}
|
|
343
|
+
const toTextSync = () => content.map(n => {
|
|
344
|
+
const getText = (node) => {
|
|
345
|
+
if (node.type === 'text' || node.type === 'code')
|
|
346
|
+
return node.text || '';
|
|
347
|
+
if (node.type === 'break')
|
|
348
|
+
return '\n';
|
|
349
|
+
if (node.children) {
|
|
350
|
+
const isBlock = ['table', 'row', 'list', 'sheet', 'slide'].includes(node.type);
|
|
351
|
+
return node.children.map(getText).join(isBlock ? config.newlineDelimiter : '');
|
|
352
|
+
}
|
|
353
|
+
return '';
|
|
354
|
+
};
|
|
355
|
+
return getText(n);
|
|
356
|
+
}).join(config.newlineDelimiter)
|
|
357
|
+
.replace(/\n{3,}/g, '\n\n'); // Normalize excessive whitespace
|
|
358
|
+
return (0, astUtils_js_1.createAST)('md', metadata, content, attachments, config, toTextSync);
|
|
359
|
+
};
|
|
360
|
+
exports.parseMarkdown = parseMarkdown;
|
|
@@ -20,7 +20,7 @@
|
|
|
20
20
|
*
|
|
21
21
|
* @module OpenOfficeParser
|
|
22
22
|
*/
|
|
23
|
-
import {
|
|
23
|
+
import { FullOfficeParserConfig, OfficeParserAST } from '../types.js';
|
|
24
24
|
/**
|
|
25
25
|
* Parses an OpenOffice document (.odt, .odp, .ods) and extracts content.
|
|
26
26
|
*
|
|
@@ -28,4 +28,4 @@ import { OfficeParserAST, OfficeParserConfig } from '../types.js';
|
|
|
28
28
|
* @param config - Parser configuration
|
|
29
29
|
* @returns A promise resolving to the parsed AST
|
|
30
30
|
*/
|
|
31
|
-
export declare const parseOpenOffice: (buffer: Buffer, config:
|
|
31
|
+
export declare const parseOpenOffice: (buffer: Buffer, config: FullOfficeParserConfig) => Promise<OfficeParserAST>;
|