officeparser 6.1.1 → 7.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +219 -26
- package/dist/OfficeConverter.d.ts +46 -0
- package/dist/OfficeConverter.js +72 -0
- package/dist/OfficeGenerator.d.ts +19 -0
- package/dist/OfficeGenerator.js +48 -0
- package/dist/OfficeParser.d.ts +6 -0
- package/dist/OfficeParser.js +55 -29
- package/dist/cli.d.ts +3 -1
- package/dist/cli.js +106 -22
- package/dist/defaults.d.ts +41 -0
- package/dist/defaults.js +172 -0
- package/dist/generators/BaseGenerator.d.ts +58 -0
- package/dist/generators/BaseGenerator.js +107 -0
- package/dist/generators/ChunkingGenerator.d.ts +81 -0
- package/dist/generators/ChunkingGenerator.js +683 -0
- package/dist/generators/CsvGenerator.d.ts +30 -0
- package/dist/generators/CsvGenerator.js +233 -0
- package/dist/generators/HtmlGenerator.d.ts +37 -0
- package/dist/generators/HtmlGenerator.js +1013 -0
- package/dist/generators/MarkdownGenerator.d.ts +59 -0
- package/dist/generators/MarkdownGenerator.js +481 -0
- package/dist/generators/PdfGenerator.d.ts +22 -0
- package/dist/generators/PdfGenerator.js +118 -0
- package/dist/generators/RtfGenerator.d.ts +15 -0
- package/dist/generators/RtfGenerator.js +208 -0
- package/dist/generators/TextGenerator.d.ts +13 -0
- package/dist/generators/TextGenerator.js +108 -0
- package/dist/index.d.ts +11 -3
- package/dist/index.js +17 -2
- package/dist/index.mjs +2 -2
- package/dist/officeparser.browser.d.ts +826 -5
- package/dist/officeparser.browser.iife.js +703 -52
- package/dist/officeparser.browser.mjs +703 -52
- package/dist/parsers/CsvParser.d.ts +9 -0
- package/dist/parsers/CsvParser.js +110 -0
- package/dist/parsers/ExcelParser.d.ts +2 -2
- package/dist/parsers/ExcelParser.js +145 -114
- package/dist/parsers/HtmlParser.d.ts +2 -0
- package/dist/parsers/HtmlParser.js +539 -0
- package/dist/parsers/MarkdownParser.d.ts +2 -0
- package/dist/parsers/MarkdownParser.js +360 -0
- package/dist/parsers/OpenOfficeParser.d.ts +2 -2
- package/dist/parsers/OpenOfficeParser.js +140 -79
- package/dist/parsers/PdfParser.d.ts +2 -2
- package/dist/parsers/PdfParser.js +52 -49
- package/dist/parsers/PowerPointParser.d.ts +2 -2
- package/dist/parsers/PowerPointParser.js +20 -23
- package/dist/parsers/RtfParser.d.ts +2 -2
- package/dist/parsers/RtfParser.js +1291 -1240
- package/dist/parsers/WordParser.d.ts +2 -2
- package/dist/parsers/WordParser.js +232 -97
- package/dist/sbom.cdx.json +99 -99
- package/dist/types.d.ts +781 -5
- package/dist/types.js +71 -0
- package/dist/utils/astUtils.d.ts +16 -0
- package/dist/utils/astUtils.js +32 -0
- package/dist/utils/configUtils.d.ts +26 -0
- package/dist/utils/configUtils.js +140 -0
- package/dist/utils/envUtils.js +56 -2
- package/dist/utils/errorUtils.d.ts +17 -29
- package/dist/utils/errorUtils.js +109 -52
- package/dist/utils/moduleLoader.js +15 -9
- package/dist/utils/ocrUtils.js +2 -1
- package/dist/utils/sheetUtils.d.ts +7 -0
- package/dist/utils/sheetUtils.js +35 -0
- package/dist/utils/styleMapper.d.ts +36 -0
- package/dist/utils/styleMapper.js +224 -0
- package/dist/utils/xmlUtils.d.ts +0 -8
- package/dist/utils/xmlUtils.js +2 -1
- package/package.json +27 -8
|
@@ -0,0 +1,539 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.parseHtml = void 0;
|
|
4
|
+
const astUtils_js_1 = require("../utils/astUtils.js");
|
|
5
|
+
const parseAttributes = (attrString) => {
|
|
6
|
+
const attrs = {};
|
|
7
|
+
const regex = /([a-zA-Z0-9\-:]+)(?:\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s>]+)))?/g;
|
|
8
|
+
let match;
|
|
9
|
+
while ((match = regex.exec(attrString)) !== null) {
|
|
10
|
+
const name = match[1].toLowerCase();
|
|
11
|
+
const value = match[2] !== undefined ? match[2] : (match[3] !== undefined ? match[3] : (match[4] || ''));
|
|
12
|
+
attrs[name] = value;
|
|
13
|
+
}
|
|
14
|
+
return attrs;
|
|
15
|
+
};
|
|
16
|
+
const parseHtmlTree = (html) => {
|
|
17
|
+
const root = { type: 'element', tagName: 'root', children: [], attributes: {} };
|
|
18
|
+
let current = root;
|
|
19
|
+
let cursor = 0;
|
|
20
|
+
while (cursor < html.length) {
|
|
21
|
+
const tagStart = html.indexOf('<', cursor);
|
|
22
|
+
if (tagStart === -1) {
|
|
23
|
+
const text = html.substring(cursor);
|
|
24
|
+
if (text)
|
|
25
|
+
current.children.push({ type: 'text', text, children: [], parent: current });
|
|
26
|
+
break;
|
|
27
|
+
}
|
|
28
|
+
if (tagStart > cursor) {
|
|
29
|
+
const text = html.substring(cursor, tagStart);
|
|
30
|
+
if (text)
|
|
31
|
+
current.children.push({ type: 'text', text, children: [], parent: current });
|
|
32
|
+
}
|
|
33
|
+
if (html.startsWith('<!--', tagStart)) {
|
|
34
|
+
const commentEnd = html.indexOf('-->', tagStart + 4);
|
|
35
|
+
cursor = commentEnd !== -1 ? commentEnd + 3 : html.length;
|
|
36
|
+
continue;
|
|
37
|
+
}
|
|
38
|
+
const tagEndMatch = html.substring(tagStart).match(/>/);
|
|
39
|
+
if (!tagEndMatch) {
|
|
40
|
+
const text = html.substring(tagStart);
|
|
41
|
+
current.children.push({ type: 'text', text, children: [], parent: current });
|
|
42
|
+
break;
|
|
43
|
+
}
|
|
44
|
+
const tagContent = html.substring(tagStart + 1, tagStart + tagEndMatch.index);
|
|
45
|
+
cursor = tagStart + tagEndMatch.index + 1;
|
|
46
|
+
const isClosing = tagContent.startsWith('/');
|
|
47
|
+
const isSelfClosing = tagContent.endsWith('/');
|
|
48
|
+
const tagCore = tagContent.replace(/^\/|\/$/g, '').trim();
|
|
49
|
+
const firstSpace = tagCore.search(/\s/);
|
|
50
|
+
const tagName = (firstSpace === -1 ? tagCore : tagCore.substring(0, firstSpace)).toLowerCase();
|
|
51
|
+
const attrString = firstSpace === -1 ? '' : tagCore.substring(firstSpace);
|
|
52
|
+
if (!tagName || !tagName.match(/^[a-z0-9\-]+$/)) {
|
|
53
|
+
// Probably not a real tag, e.g., < 5
|
|
54
|
+
current.children.push({ type: 'text', text: `<${tagContent}>`, children: [], parent: current });
|
|
55
|
+
continue;
|
|
56
|
+
}
|
|
57
|
+
if (isClosing) {
|
|
58
|
+
let p = current;
|
|
59
|
+
while (p && p.tagName !== tagName) {
|
|
60
|
+
p = p.parent;
|
|
61
|
+
}
|
|
62
|
+
if (p && p.parent) {
|
|
63
|
+
current = p.parent;
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
else {
|
|
67
|
+
const node = {
|
|
68
|
+
type: 'element',
|
|
69
|
+
tagName,
|
|
70
|
+
attributes: parseAttributes(attrString),
|
|
71
|
+
children: [],
|
|
72
|
+
parent: current
|
|
73
|
+
};
|
|
74
|
+
current.children.push(node);
|
|
75
|
+
const voidElements = new Set(['area', 'base', 'br', 'col', 'embed', 'hr', 'img', 'input', 'link', 'meta', 'param', 'source', 'track', 'wbr', '!doctype']);
|
|
76
|
+
if (!isSelfClosing && !voidElements.has(tagName)) {
|
|
77
|
+
current = node;
|
|
78
|
+
if (tagName === 'script' || tagName === 'style') {
|
|
79
|
+
const closeTag = `</${tagName}>`;
|
|
80
|
+
const closeIdx = html.toLowerCase().indexOf(closeTag, cursor);
|
|
81
|
+
if (closeIdx !== -1) {
|
|
82
|
+
node.children.push({
|
|
83
|
+
type: 'text',
|
|
84
|
+
text: html.substring(cursor, closeIdx),
|
|
85
|
+
children: [],
|
|
86
|
+
parent: node
|
|
87
|
+
});
|
|
88
|
+
cursor = closeIdx + closeTag.length;
|
|
89
|
+
current = node.parent;
|
|
90
|
+
}
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
}
|
|
94
|
+
}
|
|
95
|
+
return root;
|
|
96
|
+
};
|
|
97
|
+
const parseHtml = async (buffer, config) => {
|
|
98
|
+
const textStr = buffer.toString('utf-8');
|
|
99
|
+
const root = parseHtmlTree(textStr);
|
|
100
|
+
// Find head and body
|
|
101
|
+
let head;
|
|
102
|
+
let body = root;
|
|
103
|
+
const findNode = (node, tag) => {
|
|
104
|
+
if (node.tagName === tag)
|
|
105
|
+
return node;
|
|
106
|
+
for (const child of node.children) {
|
|
107
|
+
const found = findNode(child, tag);
|
|
108
|
+
if (found)
|
|
109
|
+
return found;
|
|
110
|
+
}
|
|
111
|
+
return undefined;
|
|
112
|
+
};
|
|
113
|
+
const htmlNode = findNode(root, 'html');
|
|
114
|
+
if (htmlNode) {
|
|
115
|
+
head = findNode(htmlNode, 'head');
|
|
116
|
+
body = findNode(htmlNode, 'body') || htmlNode;
|
|
117
|
+
}
|
|
118
|
+
const metadata = {};
|
|
119
|
+
const attachments = [];
|
|
120
|
+
if (head) {
|
|
121
|
+
const titleNode = findNode(head, 'title');
|
|
122
|
+
if (titleNode && titleNode.children.length > 0 && titleNode.children[0].text) {
|
|
123
|
+
metadata.title = titleNode.children[0].text;
|
|
124
|
+
}
|
|
125
|
+
const extractMeta = (name) => {
|
|
126
|
+
for (const child of head.children) {
|
|
127
|
+
if (child.tagName === 'meta' && (child.attributes?.name === name || child.attributes?.property === name)) {
|
|
128
|
+
return child.attributes?.content;
|
|
129
|
+
}
|
|
130
|
+
}
|
|
131
|
+
return undefined;
|
|
132
|
+
};
|
|
133
|
+
const author = extractMeta('author');
|
|
134
|
+
if (author)
|
|
135
|
+
metadata.author = author;
|
|
136
|
+
const desc = extractMeta('description');
|
|
137
|
+
if (desc)
|
|
138
|
+
metadata.description = desc;
|
|
139
|
+
const created = extractMeta('dcterms.created');
|
|
140
|
+
if (created)
|
|
141
|
+
metadata.created = new Date(created);
|
|
142
|
+
const modified = extractMeta('dcterms.modified');
|
|
143
|
+
if (modified)
|
|
144
|
+
metadata.modified = new Date(modified);
|
|
145
|
+
const lastMod = extractMeta('lastModifiedBy');
|
|
146
|
+
if (lastMod)
|
|
147
|
+
metadata.lastModifiedBy = lastMod;
|
|
148
|
+
// Custom properties
|
|
149
|
+
const customProps = {};
|
|
150
|
+
for (const child of head.children) {
|
|
151
|
+
if (child.tagName === 'meta' && child.attributes?.name?.startsWith('custom:')) {
|
|
152
|
+
const key = child.attributes.name.substring(7);
|
|
153
|
+
const val = child.attributes.content || '';
|
|
154
|
+
// Try to infer type
|
|
155
|
+
if (val === 'true')
|
|
156
|
+
customProps[key] = true;
|
|
157
|
+
else if (val === 'false')
|
|
158
|
+
customProps[key] = false;
|
|
159
|
+
else if (!isNaN(Number(val)) && val.trim() !== '')
|
|
160
|
+
customProps[key] = Number(val);
|
|
161
|
+
else if (!isNaN(Date.parse(val)) && val.includes(':'))
|
|
162
|
+
customProps[key] = new Date(val);
|
|
163
|
+
else
|
|
164
|
+
customProps[key] = val;
|
|
165
|
+
}
|
|
166
|
+
}
|
|
167
|
+
if (Object.keys(customProps).length > 0)
|
|
168
|
+
metadata.customProperties = customProps;
|
|
169
|
+
}
|
|
170
|
+
const content = [];
|
|
171
|
+
let htmlListIdCounter = 1;
|
|
172
|
+
const parseNode = (node, currentFormatting = {}, listContext) => {
|
|
173
|
+
if (node.type === 'text') {
|
|
174
|
+
let decodedText = (node.text || '')
|
|
175
|
+
.replace(/ /g, ' ')
|
|
176
|
+
.replace(/</g, '<')
|
|
177
|
+
.replace(/>/g, '>')
|
|
178
|
+
.replace(/&/g, '&')
|
|
179
|
+
.replace(/"/g, '"')
|
|
180
|
+
.replace(/'/g, "'");
|
|
181
|
+
if (!config.preserveXmlWhitespace) {
|
|
182
|
+
decodedText = decodedText.replace(/\s+/g, ' ');
|
|
183
|
+
}
|
|
184
|
+
if (!decodedText.trim() && !config.preserveXmlWhitespace)
|
|
185
|
+
return null;
|
|
186
|
+
const textNode = {
|
|
187
|
+
type: 'text',
|
|
188
|
+
text: decodedText,
|
|
189
|
+
formatting: Object.keys(currentFormatting).length > 0 ? { ...currentFormatting } : undefined
|
|
190
|
+
};
|
|
191
|
+
if (config.includeRawContent && node.text) {
|
|
192
|
+
// For text nodes in this manual parser, we just use the decoded text as raw content
|
|
193
|
+
// as we don't have accurate locators for the original source slice
|
|
194
|
+
textNode.rawContent = node.text;
|
|
195
|
+
}
|
|
196
|
+
return textNode;
|
|
197
|
+
}
|
|
198
|
+
if (node.type === 'element' && node.tagName) {
|
|
199
|
+
const tagName = node.tagName;
|
|
200
|
+
const newFormatting = { ...currentFormatting };
|
|
201
|
+
if (tagName === 'b' || tagName === 'strong')
|
|
202
|
+
newFormatting.bold = true;
|
|
203
|
+
if (tagName === 'i' || tagName === 'em')
|
|
204
|
+
newFormatting.italic = true;
|
|
205
|
+
if (tagName === 'u')
|
|
206
|
+
newFormatting.underline = true;
|
|
207
|
+
if (tagName === 'strike' || tagName === 's' || tagName === 'del')
|
|
208
|
+
newFormatting.strikethrough = true;
|
|
209
|
+
if (tagName === 'sub')
|
|
210
|
+
newFormatting.subscript = true;
|
|
211
|
+
if (tagName === 'sup')
|
|
212
|
+
newFormatting.superscript = true;
|
|
213
|
+
if (tagName === 'code')
|
|
214
|
+
newFormatting.font = 'monospace';
|
|
215
|
+
const styleAttr = node.attributes?.style || '';
|
|
216
|
+
const alignAttr = node.attributes?.align || '';
|
|
217
|
+
if (styleAttr || alignAttr) {
|
|
218
|
+
if (styleAttr.includes('font-weight: bold'))
|
|
219
|
+
newFormatting.bold = true;
|
|
220
|
+
if (styleAttr.includes('font-style: italic'))
|
|
221
|
+
newFormatting.italic = true;
|
|
222
|
+
if (styleAttr.includes('text-decoration: underline'))
|
|
223
|
+
newFormatting.underline = true;
|
|
224
|
+
if (styleAttr.includes('text-decoration: line-through'))
|
|
225
|
+
newFormatting.strikethrough = true;
|
|
226
|
+
const colorMatch = styleAttr.match(/color:\s*([^;]+)/);
|
|
227
|
+
if (colorMatch)
|
|
228
|
+
newFormatting.color = colorMatch[1].trim();
|
|
229
|
+
const bgMatch = styleAttr.match(/background-color:\s*([^;]+)/);
|
|
230
|
+
if (bgMatch)
|
|
231
|
+
newFormatting.backgroundColor = bgMatch[1].trim();
|
|
232
|
+
const sizeMatch = styleAttr.match(/font-size:\s*([^;]+)/);
|
|
233
|
+
if (sizeMatch)
|
|
234
|
+
newFormatting.size = sizeMatch[1].trim();
|
|
235
|
+
const fontMatch = styleAttr.match(/font-family:\s*([^;]+)/);
|
|
236
|
+
if (fontMatch)
|
|
237
|
+
newFormatting.font = fontMatch[1].trim().split(',')[0].replace(/['"]/g, '');
|
|
238
|
+
const alignmentMatch = styleAttr.match(/text-align:\s*(left|center|right|justify)/);
|
|
239
|
+
if (alignmentMatch) {
|
|
240
|
+
newFormatting.alignment = alignmentMatch[1].toLowerCase();
|
|
241
|
+
}
|
|
242
|
+
else if (alignAttr) {
|
|
243
|
+
const align = alignAttr.toLowerCase();
|
|
244
|
+
if (['left', 'center', 'right', 'justify'].includes(align)) {
|
|
245
|
+
newFormatting.alignment = align;
|
|
246
|
+
}
|
|
247
|
+
}
|
|
248
|
+
}
|
|
249
|
+
const anchorIds = node.attributes?.id ? [node.attributes.id] : [];
|
|
250
|
+
const parseChildren = (n, fmt, lCtx) => {
|
|
251
|
+
const kids = [];
|
|
252
|
+
for (const child of n.children) {
|
|
253
|
+
const parsed = parseNode(child, fmt, lCtx);
|
|
254
|
+
if (parsed) {
|
|
255
|
+
if (Array.isArray(parsed))
|
|
256
|
+
kids.push(...parsed);
|
|
257
|
+
else
|
|
258
|
+
kids.push(parsed);
|
|
259
|
+
}
|
|
260
|
+
}
|
|
261
|
+
return kids;
|
|
262
|
+
};
|
|
263
|
+
// Skip structural containers produced by HtmlGenerator to avoid deep AST nesting
|
|
264
|
+
if (tagName === 'div' && (node.attributes?.class === 'container' ||
|
|
265
|
+
node.attributes?.class === 'spreadsheet-container' ||
|
|
266
|
+
node.attributes?.class === 'presentation-container' ||
|
|
267
|
+
node.attributes?.class === 'pdf-container' ||
|
|
268
|
+
node.attributes?.class === 'metadata-summary' ||
|
|
269
|
+
node.attributes?.class === 'image-container' ||
|
|
270
|
+
node.attributes?.class === 'chart-container' ||
|
|
271
|
+
node.attributes?.class === 'table-container' ||
|
|
272
|
+
node.attributes?.class === 'caption' ||
|
|
273
|
+
node.attributes?.class === 'sheet' ||
|
|
274
|
+
node.attributes?.class === 'page' ||
|
|
275
|
+
node.attributes?.class === 'slide' ||
|
|
276
|
+
node.attributes?.class === 'note-content')) {
|
|
277
|
+
return parseChildren(node, newFormatting, listContext);
|
|
278
|
+
}
|
|
279
|
+
if (tagName === 'article') {
|
|
280
|
+
return parseChildren(node, newFormatting, listContext);
|
|
281
|
+
}
|
|
282
|
+
if (tagName === 'p' || tagName === 'div') {
|
|
283
|
+
const children = parseChildren(node, newFormatting, listContext);
|
|
284
|
+
// If it's a div and contains block elements, return children directly
|
|
285
|
+
const hasBlockElements = children.some(c => ['paragraph', 'table', 'heading', 'list', 'image', 'chart', 'code'].includes(c.type));
|
|
286
|
+
if (tagName === 'div' && hasBlockElements) {
|
|
287
|
+
return children;
|
|
288
|
+
}
|
|
289
|
+
// Flatten nested paragraphs to avoid deep AST nesting (e.g. from notes)
|
|
290
|
+
const flattenedChildren = [];
|
|
291
|
+
for (const child of children) {
|
|
292
|
+
if (child.type === 'paragraph' && child.children) {
|
|
293
|
+
flattenedChildren.push(...child.children);
|
|
294
|
+
}
|
|
295
|
+
else {
|
|
296
|
+
flattenedChildren.push(child);
|
|
297
|
+
}
|
|
298
|
+
}
|
|
299
|
+
const pNode = {
|
|
300
|
+
type: 'paragraph',
|
|
301
|
+
metadata: { alignment: newFormatting.alignment, anchorIds: anchorIds.length > 0 ? anchorIds : undefined },
|
|
302
|
+
children: flattenedChildren
|
|
303
|
+
};
|
|
304
|
+
if (config.includeRawContent) {
|
|
305
|
+
// Note: Since this is a manual parser without locators, we can't easily get the original source slice.
|
|
306
|
+
// We'll skip rawContent for structural nodes here unless we want to implement index tracking in parseHtmlTree.
|
|
307
|
+
}
|
|
308
|
+
return pNode;
|
|
309
|
+
}
|
|
310
|
+
if (tagName.match(/^h[1-6]$/)) {
|
|
311
|
+
const level = parseInt(tagName.substring(1));
|
|
312
|
+
const hNode = {
|
|
313
|
+
type: 'heading',
|
|
314
|
+
metadata: { level, alignment: newFormatting.alignment, anchorIds: anchorIds.length > 0 ? anchorIds : undefined },
|
|
315
|
+
children: parseChildren(node, newFormatting, listContext)
|
|
316
|
+
};
|
|
317
|
+
return hNode;
|
|
318
|
+
}
|
|
319
|
+
if (tagName === 'ul' || tagName === 'ol') {
|
|
320
|
+
const isNewTopLevel = !listContext;
|
|
321
|
+
const newListContext = {
|
|
322
|
+
listId: isNewTopLevel ? `html-list-${htmlListIdCounter++}` : listContext.listId,
|
|
323
|
+
type: tagName === 'ol' ? 'ordered' : 'unordered',
|
|
324
|
+
level: isNewTopLevel ? 0 : listContext.level + 1,
|
|
325
|
+
counters: isNewTopLevel ? {} : { ...listContext.counters } // Clone to avoid side effects on parent levels
|
|
326
|
+
};
|
|
327
|
+
// Initialize counter for this level
|
|
328
|
+
if (tagName === 'ol' && node.attributes?.start) {
|
|
329
|
+
const start = parseInt(node.attributes.start, 10);
|
|
330
|
+
newListContext.counters[newListContext.level] = isNaN(start) ? 0 : start - 1;
|
|
331
|
+
}
|
|
332
|
+
else {
|
|
333
|
+
newListContext.counters[newListContext.level] = 0;
|
|
334
|
+
}
|
|
335
|
+
return parseChildren(node, currentFormatting, newListContext);
|
|
336
|
+
}
|
|
337
|
+
if (tagName === 'li') {
|
|
338
|
+
if (listContext) {
|
|
339
|
+
if (node.attributes?.value) {
|
|
340
|
+
const val = parseInt(node.attributes.value, 10);
|
|
341
|
+
if (!isNaN(val))
|
|
342
|
+
listContext.counters[listContext.level] = val;
|
|
343
|
+
}
|
|
344
|
+
else {
|
|
345
|
+
listContext.counters[listContext.level]++;
|
|
346
|
+
}
|
|
347
|
+
}
|
|
348
|
+
const children = parseChildren(node, newFormatting, listContext);
|
|
349
|
+
const nestedLists = children.filter(c => c.type === 'list');
|
|
350
|
+
const selfChildren = children.filter(c => c.type !== 'list');
|
|
351
|
+
const selfNode = {
|
|
352
|
+
type: 'list',
|
|
353
|
+
text: selfChildren.map(c => c.text || '').join(''),
|
|
354
|
+
metadata: {
|
|
355
|
+
listType: listContext?.type || 'unordered',
|
|
356
|
+
indentation: listContext?.level || 0,
|
|
357
|
+
alignment: newFormatting.alignment || 'left',
|
|
358
|
+
listId: listContext?.listId || 'html-list-none',
|
|
359
|
+
itemIndex: (listContext?.counters[listContext.level] ?? 1) - 1,
|
|
360
|
+
anchorIds: anchorIds.length > 0 ? anchorIds : undefined
|
|
361
|
+
},
|
|
362
|
+
children: selfChildren
|
|
363
|
+
};
|
|
364
|
+
return [selfNode, ...nestedLists];
|
|
365
|
+
}
|
|
366
|
+
if (tagName === 'table') {
|
|
367
|
+
const tableNode = {
|
|
368
|
+
type: 'table',
|
|
369
|
+
metadata: { anchorIds: anchorIds.length > 0 ? anchorIds : undefined },
|
|
370
|
+
children: parseChildren(node, newFormatting, listContext)
|
|
371
|
+
};
|
|
372
|
+
if (config.includeRawContent) {
|
|
373
|
+
tableNode.rawContent = '<table>...</table>';
|
|
374
|
+
}
|
|
375
|
+
return tableNode;
|
|
376
|
+
}
|
|
377
|
+
if (tagName === 'tr') {
|
|
378
|
+
const rowNode = {
|
|
379
|
+
type: 'row',
|
|
380
|
+
children: parseChildren(node, newFormatting, listContext)
|
|
381
|
+
};
|
|
382
|
+
if (config.includeRawContent) {
|
|
383
|
+
rowNode.rawContent = '<tr>...</tr>';
|
|
384
|
+
}
|
|
385
|
+
return rowNode;
|
|
386
|
+
}
|
|
387
|
+
if (tagName === 'td' || tagName === 'th') {
|
|
388
|
+
const cellNode = {
|
|
389
|
+
type: 'cell',
|
|
390
|
+
children: parseChildren(node, newFormatting, listContext)
|
|
391
|
+
};
|
|
392
|
+
if (config.includeRawContent) {
|
|
393
|
+
cellNode.rawContent = '<td>...</td>';
|
|
394
|
+
}
|
|
395
|
+
return cellNode;
|
|
396
|
+
}
|
|
397
|
+
if (tagName === 'img') {
|
|
398
|
+
const src = node.attributes?.src;
|
|
399
|
+
const alt = node.attributes?.alt;
|
|
400
|
+
let imageNode;
|
|
401
|
+
if (src?.startsWith('data:')) {
|
|
402
|
+
const match = src.match(/^data:([^;]+);base64,(.*)$/);
|
|
403
|
+
if (match && config.extractAttachments) {
|
|
404
|
+
const mimeType = match[1];
|
|
405
|
+
const data = match[2];
|
|
406
|
+
const name = `image_${attachments.length + 1}.${mimeType.split('/')[1]}`;
|
|
407
|
+
attachments.push({
|
|
408
|
+
type: 'image',
|
|
409
|
+
mimeType,
|
|
410
|
+
data,
|
|
411
|
+
name,
|
|
412
|
+
extension: mimeType.split('/')[1]
|
|
413
|
+
});
|
|
414
|
+
imageNode = {
|
|
415
|
+
type: 'image',
|
|
416
|
+
metadata: {
|
|
417
|
+
attachmentName: name,
|
|
418
|
+
altText: alt
|
|
419
|
+
}
|
|
420
|
+
};
|
|
421
|
+
}
|
|
422
|
+
else {
|
|
423
|
+
imageNode = {
|
|
424
|
+
type: 'image',
|
|
425
|
+
metadata: {
|
|
426
|
+
url: src,
|
|
427
|
+
altText: alt
|
|
428
|
+
}
|
|
429
|
+
};
|
|
430
|
+
}
|
|
431
|
+
}
|
|
432
|
+
else {
|
|
433
|
+
imageNode = {
|
|
434
|
+
type: 'image',
|
|
435
|
+
metadata: {
|
|
436
|
+
url: src,
|
|
437
|
+
altText: alt,
|
|
438
|
+
anchorIds: anchorIds.length > 0 ? anchorIds : undefined
|
|
439
|
+
}
|
|
440
|
+
};
|
|
441
|
+
}
|
|
442
|
+
if (config.includeRawContent) {
|
|
443
|
+
imageNode.rawContent = '<img>';
|
|
444
|
+
}
|
|
445
|
+
return imageNode;
|
|
446
|
+
}
|
|
447
|
+
if (tagName === 'a') {
|
|
448
|
+
const href = node.attributes?.href;
|
|
449
|
+
const children = parseChildren(node, newFormatting, listContext);
|
|
450
|
+
if (href) {
|
|
451
|
+
const linkType = href.startsWith('#') ? 'internal' : 'external';
|
|
452
|
+
children.forEach(c => {
|
|
453
|
+
if (c.type === 'text') {
|
|
454
|
+
c.metadata = { ...c.metadata, link: href, linkType };
|
|
455
|
+
}
|
|
456
|
+
});
|
|
457
|
+
}
|
|
458
|
+
return children;
|
|
459
|
+
}
|
|
460
|
+
if (tagName === 'br') {
|
|
461
|
+
const brNode = { type: 'break', metadata: { breakType: 'textWrapping' } };
|
|
462
|
+
if (config.includeRawContent) {
|
|
463
|
+
brNode.rawContent = '<br/>';
|
|
464
|
+
}
|
|
465
|
+
return brNode;
|
|
466
|
+
}
|
|
467
|
+
if (tagName === 'pre') {
|
|
468
|
+
const codeNode = node.children.find(c => c.tagName === 'code');
|
|
469
|
+
let language;
|
|
470
|
+
let codeText = '';
|
|
471
|
+
if (codeNode) {
|
|
472
|
+
const classAttr = codeNode.attributes?.class || '';
|
|
473
|
+
const langMatch = classAttr.split(' ').find((c) => c.startsWith('language-'));
|
|
474
|
+
if (langMatch)
|
|
475
|
+
language = langMatch.replace('language-', '');
|
|
476
|
+
codeText = codeNode.children.map(c => c.text || '').join('');
|
|
477
|
+
}
|
|
478
|
+
else {
|
|
479
|
+
codeText = node.children.map(c => c.text || '').join('');
|
|
480
|
+
}
|
|
481
|
+
const preNode = {
|
|
482
|
+
type: 'code',
|
|
483
|
+
text: codeText,
|
|
484
|
+
metadata: { language, anchorIds: anchorIds.length > 0 ? anchorIds : undefined }
|
|
485
|
+
};
|
|
486
|
+
if (config.includeRawContent) {
|
|
487
|
+
preNode.rawContent = '<pre>...</pre>';
|
|
488
|
+
}
|
|
489
|
+
return preNode;
|
|
490
|
+
}
|
|
491
|
+
if (tagName === 'script' || tagName === 'style' || tagName === '!doctype') {
|
|
492
|
+
return null;
|
|
493
|
+
}
|
|
494
|
+
return parseChildren(node, newFormatting, listContext);
|
|
495
|
+
}
|
|
496
|
+
return null;
|
|
497
|
+
};
|
|
498
|
+
for (const child of body.children) {
|
|
499
|
+
const parsed = parseNode(child);
|
|
500
|
+
if (parsed) {
|
|
501
|
+
if (Array.isArray(parsed)) {
|
|
502
|
+
parsed.forEach(p => {
|
|
503
|
+
if (p.type === 'text') {
|
|
504
|
+
// Wrap direct body text in paragraphs
|
|
505
|
+
content.push({ type: 'paragraph', children: [p] });
|
|
506
|
+
}
|
|
507
|
+
else {
|
|
508
|
+
content.push(p);
|
|
509
|
+
}
|
|
510
|
+
});
|
|
511
|
+
}
|
|
512
|
+
else {
|
|
513
|
+
if (parsed.type === 'text') {
|
|
514
|
+
content.push({ type: 'paragraph', children: [parsed] });
|
|
515
|
+
}
|
|
516
|
+
else {
|
|
517
|
+
content.push(parsed);
|
|
518
|
+
}
|
|
519
|
+
}
|
|
520
|
+
}
|
|
521
|
+
}
|
|
522
|
+
const toTextSync = () => content.map(n => {
|
|
523
|
+
const getText = (node) => {
|
|
524
|
+
if (node.type === 'text' || node.type === 'code')
|
|
525
|
+
return node.text || '';
|
|
526
|
+
if (node.type === 'break')
|
|
527
|
+
return '\n';
|
|
528
|
+
if (node.children) {
|
|
529
|
+
const isBlock = ['table', 'row', 'list', 'sheet', 'slide'].includes(node.type);
|
|
530
|
+
return node.children.map(getText).join(isBlock ? config.newlineDelimiter : '');
|
|
531
|
+
}
|
|
532
|
+
return '';
|
|
533
|
+
};
|
|
534
|
+
return getText(n);
|
|
535
|
+
}).join(config.newlineDelimiter)
|
|
536
|
+
.replace(/\n{3,}/g, '\n\n'); // Normalize excessive whitespace
|
|
537
|
+
return (0, astUtils_js_1.createAST)('html', metadata, content, attachments, config, toTextSync);
|
|
538
|
+
};
|
|
539
|
+
exports.parseHtml = parseHtml;
|