officeparser 7.2.2 → 7.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +161 -17
- package/dist/OfficeGenerator.js +4 -0
- package/dist/OfficeParser.d.ts +2 -0
- package/dist/OfficeParser.js +6 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +3 -2
- package/dist/defaults.js +3 -3
- package/dist/generators/BaseGenerator.d.ts +11 -0
- package/dist/generators/BaseGenerator.js +29 -0
- package/dist/generators/CsvGenerator.d.ts +9 -1
- package/dist/generators/CsvGenerator.js +24 -14
- package/dist/generators/EpubGenerator.d.ts +18 -0
- package/dist/generators/EpubGenerator.js +242 -0
- package/dist/generators/HtmlGenerator.d.ts +12 -0
- package/dist/generators/HtmlGenerator.js +266 -51
- package/dist/generators/MarkdownGenerator.d.ts +16 -0
- package/dist/generators/MarkdownGenerator.js +173 -24
- package/dist/generators/PdfGenerator.js +32 -0
- package/dist/generators/RtfGenerator.js +12 -15
- package/dist/generators/TextGenerator.js +11 -0
- package/dist/index.d.ts +1 -0
- package/dist/index.js +1 -0
- package/dist/officeparser.browser.d.ts +144 -7
- package/dist/officeparser.browser.iife.js +289 -193
- package/dist/officeparser.browser.mjs +289 -193
- package/dist/officeparser.browser.slim.d.ts +2129 -0
- package/dist/officeparser.browser.slim.iife.js +1278 -0
- package/dist/officeparser.browser.slim.mjs +1277 -0
- package/dist/parsers/EpubParser.d.ts +8 -0
- package/dist/parsers/EpubParser.js +217 -0
- package/dist/parsers/HtmlParser.js +284 -20
- package/dist/parsers/MarkdownParser.js +424 -33
- package/dist/parsers/OpenOfficeParser.js +241 -54
- package/dist/parsers/PdfParser.js +4 -1
- package/dist/parsers/WordParser.js +2 -2
- package/dist/sbom.cdx.json +111 -223
- package/dist/types.d.ts +146 -7
- package/dist/types.js +2 -0
- package/dist/utils/errorUtils.js +3 -2
- package/dist/utils/sanitize.d.ts +99 -0
- package/dist/utils/sanitize.js +228 -0
- package/dist/utils/zipUtils.js +76 -26
- package/package.json +19 -10
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
3
|
exports.parseHtml = void 0;
|
|
4
|
+
const types_js_1 = require("../types.js");
|
|
4
5
|
const astUtils_js_1 = require("../utils/astUtils.js");
|
|
5
6
|
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
6
7
|
const parseAttributes = (attrString) => {
|
|
@@ -36,14 +37,16 @@ const parseHtmlTree = (html) => {
|
|
|
36
37
|
cursor = commentEnd !== -1 ? commentEnd + 3 : html.length;
|
|
37
38
|
continue;
|
|
38
39
|
}
|
|
39
|
-
|
|
40
|
-
|
|
40
|
+
// indexOf (not substring().match) so scanning for the tag end is O(1) in
|
|
41
|
+
// allocation — a document with many "<" chars would otherwise be O(n^2).
|
|
42
|
+
const tagEndIdx = html.indexOf('>', tagStart);
|
|
43
|
+
if (tagEndIdx === -1) {
|
|
41
44
|
const text = html.substring(tagStart);
|
|
42
45
|
current.children.push({ type: 'text', text, children: [], parent: current });
|
|
43
46
|
break;
|
|
44
47
|
}
|
|
45
|
-
const tagContent = html.substring(tagStart + 1,
|
|
46
|
-
cursor =
|
|
48
|
+
const tagContent = html.substring(tagStart + 1, tagEndIdx);
|
|
49
|
+
cursor = tagEndIdx + 1;
|
|
47
50
|
const isClosing = tagContent.startsWith('/');
|
|
48
51
|
const isSelfClosing = tagContent.endsWith('/');
|
|
49
52
|
const tagCore = tagContent.replace(/^\/|\/$/g, '').trim();
|
|
@@ -77,16 +80,20 @@ const parseHtmlTree = (html) => {
|
|
|
77
80
|
if (!isSelfClosing && !voidElements.has(tagName)) {
|
|
78
81
|
current = node;
|
|
79
82
|
if (tagName === 'script' || tagName === 'style') {
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
+
// Case-insensitive search from `cursor` via a sticky-ish regex, instead of
|
|
84
|
+
// lower-casing the whole document on every <script>/<style> (was O(n^2)).
|
|
85
|
+
// tagName is validated to /^[a-z0-9-]+$/ above, so it's safe to interpolate.
|
|
86
|
+
const closeRe = new RegExp(`</${tagName}>`, 'gi');
|
|
87
|
+
closeRe.lastIndex = cursor;
|
|
88
|
+
const closeMatch = closeRe.exec(html);
|
|
89
|
+
if (closeMatch) {
|
|
83
90
|
node.children.push({
|
|
84
91
|
type: 'text',
|
|
85
|
-
text: html.substring(cursor,
|
|
92
|
+
text: html.substring(cursor, closeMatch.index),
|
|
86
93
|
children: [],
|
|
87
94
|
parent: node
|
|
88
95
|
});
|
|
89
|
-
cursor =
|
|
96
|
+
cursor = closeMatch.index + closeMatch[0].length;
|
|
90
97
|
current = node.parent;
|
|
91
98
|
}
|
|
92
99
|
}
|
|
@@ -183,7 +190,30 @@ const parseHtml = async (buffer, config) => {
|
|
|
183
190
|
}
|
|
184
191
|
const content = [];
|
|
185
192
|
let htmlListIdCounter = 1;
|
|
186
|
-
|
|
193
|
+
// Finds the checked state from a nested <input type="checkbox"> (GFM task-list items
|
|
194
|
+
// nest it inside a <label>, so it isn't a direct child of the <li>).
|
|
195
|
+
const findNestedCheckboxChecked = (n) => {
|
|
196
|
+
if (n.tagName === 'input' && (n.attributes?.type || '').toLowerCase() === 'checkbox') {
|
|
197
|
+
return 'checked' in (n.attributes || {});
|
|
198
|
+
}
|
|
199
|
+
for (const child of n.children) {
|
|
200
|
+
const found = findNestedCheckboxChecked(child);
|
|
201
|
+
if (found !== undefined)
|
|
202
|
+
return found;
|
|
203
|
+
}
|
|
204
|
+
return undefined;
|
|
205
|
+
};
|
|
206
|
+
// Populated from a <section data-footnotes> block (found and parsed before the main
|
|
207
|
+
// body loop, since references can appear anywhere earlier in the document) and
|
|
208
|
+
// consulted by parseChildren's <sup data-footnote-ref> handling below.
|
|
209
|
+
const footnoteDefinitions = new Map();
|
|
210
|
+
const parseNode = (node, currentFormatting = {}, listContext, depth = 0) => {
|
|
211
|
+
// Guard against a maliciously deep element tree (e.g. tens of thousands of
|
|
212
|
+
// nested <div>) recursing until the call stack overflows. Real documents
|
|
213
|
+
// nest only a few dozen levels; this trips well before a RangeError.
|
|
214
|
+
if (depth > 1000) {
|
|
215
|
+
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.MAX_NESTING_DEPTH_EXCEEDED);
|
|
216
|
+
}
|
|
187
217
|
if (node.type === 'text') {
|
|
188
218
|
let decodedText = (node.text || '')
|
|
189
219
|
.replace(/ /g, ' ')
|
|
@@ -264,7 +294,29 @@ const parseHtml = async (buffer, config) => {
|
|
|
264
294
|
const parseChildren = (n, fmt, lCtx) => {
|
|
265
295
|
const kids = [];
|
|
266
296
|
for (const child of n.children) {
|
|
267
|
-
|
|
297
|
+
// Footnote/endnote reference: attach as .notes on the preceding node
|
|
298
|
+
// instead of inserting a visible node, matching WordParser's convention.
|
|
299
|
+
if (child.type === 'element' && child.tagName === 'sup' && child.attributes?.['data-footnote-ref'] !== undefined) {
|
|
300
|
+
const key = child.attributes['data-footnote-ref'];
|
|
301
|
+
const definition = footnoteDefinitions.get(key);
|
|
302
|
+
const noteNode = {
|
|
303
|
+
type: 'note',
|
|
304
|
+
text: (definition || []).map(d => d.text || '').join(''),
|
|
305
|
+
children: definition || [],
|
|
306
|
+
metadata: { noteType: 'footnote', noteId: key }
|
|
307
|
+
};
|
|
308
|
+
if (kids.length > 0) {
|
|
309
|
+
const target = kids[kids.length - 1];
|
|
310
|
+
if (!target.notes)
|
|
311
|
+
target.notes = [];
|
|
312
|
+
target.notes.push(noteNode);
|
|
313
|
+
}
|
|
314
|
+
else {
|
|
315
|
+
kids.push({ type: 'text', text: '', notes: [noteNode] });
|
|
316
|
+
}
|
|
317
|
+
continue;
|
|
318
|
+
}
|
|
319
|
+
const parsed = parseNode(child, fmt, lCtx, depth + 1);
|
|
268
320
|
if (parsed) {
|
|
269
321
|
if (Array.isArray(parsed))
|
|
270
322
|
kids.push(...parsed);
|
|
@@ -274,6 +326,94 @@ const parseHtml = async (buffer, config) => {
|
|
|
274
326
|
}
|
|
275
327
|
return kids;
|
|
276
328
|
};
|
|
329
|
+
// YouTube embeds: inscript-editor's Youtube node renders
|
|
330
|
+
// <div data-youtube-video="ID" data-width="…" data-align="…">…<iframe…></div>.
|
|
331
|
+
// Recognise both the wrapper div and a bare iframe so externally-authored HTML
|
|
332
|
+
// (and a saved-then-reopened .md that fell back to raw HTML) both round-trip.
|
|
333
|
+
if (tagName === 'div' && node.attributes?.['data-youtube-video'] !== undefined) {
|
|
334
|
+
const videoId = node.attributes['data-youtube-video'] || '';
|
|
335
|
+
const width = node.attributes?.['data-width'];
|
|
336
|
+
const embedAlignAttr = node.attributes?.['data-align'];
|
|
337
|
+
const embedAlign = ['left', 'center', 'right'].includes(embedAlignAttr) ? embedAlignAttr : undefined;
|
|
338
|
+
const embedUrl = videoId ? `https://www.youtube.com/watch?v=${videoId}` : undefined;
|
|
339
|
+
const embedNode = {
|
|
340
|
+
type: 'embed',
|
|
341
|
+
// Childless nodes need .text so generic AST consumers (toText, chunking)
|
|
342
|
+
// don't silently drop them.
|
|
343
|
+
text: embedUrl,
|
|
344
|
+
metadata: {
|
|
345
|
+
embedType: 'youtube',
|
|
346
|
+
videoId,
|
|
347
|
+
url: embedUrl,
|
|
348
|
+
width,
|
|
349
|
+
align: embedAlign
|
|
350
|
+
}
|
|
351
|
+
};
|
|
352
|
+
if (config.includeRawContent)
|
|
353
|
+
embedNode.rawContent = '<div data-youtube-video>...</div>';
|
|
354
|
+
return embedNode;
|
|
355
|
+
}
|
|
356
|
+
if (tagName === 'iframe') {
|
|
357
|
+
const src = node.attributes?.src || '';
|
|
358
|
+
const ytMatch = /youtube(?:-nocookie)?\.com/.test(src) ? src.match(/(?:embed\/|v=)([^&?/\s]+)/) : null;
|
|
359
|
+
if (ytMatch) {
|
|
360
|
+
const embedUrl = `https://www.youtube.com/watch?v=${ytMatch[1]}`;
|
|
361
|
+
const embedNode = {
|
|
362
|
+
type: 'embed',
|
|
363
|
+
text: embedUrl,
|
|
364
|
+
metadata: { embedType: 'youtube', videoId: ytMatch[1], url: embedUrl }
|
|
365
|
+
};
|
|
366
|
+
if (config.includeRawContent)
|
|
367
|
+
embedNode.rawContent = '<iframe>...</iframe>';
|
|
368
|
+
return embedNode;
|
|
369
|
+
}
|
|
370
|
+
return null;
|
|
371
|
+
}
|
|
372
|
+
// Footnotes section: its definitions were already extracted up front (see
|
|
373
|
+
// footnoteDefinitions below), so skip it here wherever it appears in the tree -
|
|
374
|
+
// it isn't necessarily a direct child of <body> (e.g. it may be nested inside
|
|
375
|
+
// a non-standalone HtmlGenerator output's wrapping <div>).
|
|
376
|
+
if (tagName === 'section' && node.attributes?.['data-footnotes'] !== undefined) {
|
|
377
|
+
return null;
|
|
378
|
+
}
|
|
379
|
+
// Math: proposed contract (no editor node built yet) - HtmlGenerator emits
|
|
380
|
+
// <span/div class="math math-inline|math-block" data-math="inline|block">
|
|
381
|
+
// with the $-delimited LaTeX as the visible (escaped) text content.
|
|
382
|
+
if ((tagName === 'div' || tagName === 'span') && node.attributes?.['data-math'] !== undefined) {
|
|
383
|
+
const mathMode = node.attributes['data-math'] === 'block' ? 'block' : 'inline';
|
|
384
|
+
const rawText = node.children.map(c => c.text || '').join('')
|
|
385
|
+
.replace(/ /g, ' ')
|
|
386
|
+
.replace(/</g, '<')
|
|
387
|
+
.replace(/>/g, '>')
|
|
388
|
+
.replace(/&/g, '&')
|
|
389
|
+
.replace(/"/g, '"')
|
|
390
|
+
.replace(/'/g, '\'');
|
|
391
|
+
const delimiter = mathMode === 'block' ? '$$' : '$';
|
|
392
|
+
const latex = rawText.startsWith(delimiter) && rawText.endsWith(delimiter)
|
|
393
|
+
? rawText.slice(delimiter.length, -delimiter.length)
|
|
394
|
+
: rawText;
|
|
395
|
+
return {
|
|
396
|
+
type: 'code',
|
|
397
|
+
text: latex,
|
|
398
|
+
metadata: { math: mathMode }
|
|
399
|
+
};
|
|
400
|
+
}
|
|
401
|
+
// Admonition: inscript-editor's Admonition node renders
|
|
402
|
+
// <div class="admonition admonition-note" data-type="note">…children…</div>.
|
|
403
|
+
if (tagName === 'div' && (node.attributes?.class || '').split(/\s+/).includes('admonition')) {
|
|
404
|
+
const admonitionTypeAttr = node.attributes?.['data-type'];
|
|
405
|
+
const admonitionType = ['note', 'tip', 'important', 'warning', 'caution'].includes(admonitionTypeAttr)
|
|
406
|
+
? admonitionTypeAttr
|
|
407
|
+
: 'note';
|
|
408
|
+
const admonitionNode = {
|
|
409
|
+
type: 'admonition',
|
|
410
|
+
metadata: { admonitionType },
|
|
411
|
+
children: parseChildren(node, newFormatting, listContext)
|
|
412
|
+
};
|
|
413
|
+
if (config.includeRawContent)
|
|
414
|
+
admonitionNode.rawContent = '<div class="admonition">...</div>';
|
|
415
|
+
return admonitionNode;
|
|
416
|
+
}
|
|
277
417
|
// Skip structural containers produced by HtmlGenerator to avoid deep AST nesting
|
|
278
418
|
if (tagName === 'div' && (node.attributes?.class === 'container' ||
|
|
279
419
|
node.attributes?.class === 'spreadsheet-container' ||
|
|
@@ -296,7 +436,7 @@ const parseHtml = async (buffer, config) => {
|
|
|
296
436
|
if (tagName === 'p' || tagName === 'div') {
|
|
297
437
|
const children = parseChildren(node, newFormatting, listContext);
|
|
298
438
|
// If it's a div and contains block elements, return children directly
|
|
299
|
-
const hasBlockElements = children.some(c => ['paragraph', 'table', 'heading', 'list', 'image', 'chart', 'code'].includes(c.type));
|
|
439
|
+
const hasBlockElements = children.some(c => ['paragraph', 'table', 'heading', 'list', 'image', 'chart', 'code', 'embed', 'admonition', 'definitionList'].includes(c.type));
|
|
300
440
|
if (tagName === 'div' && hasBlockElements) {
|
|
301
441
|
return children;
|
|
302
442
|
}
|
|
@@ -330,13 +470,53 @@ const parseHtml = async (buffer, config) => {
|
|
|
330
470
|
};
|
|
331
471
|
return hNode;
|
|
332
472
|
}
|
|
473
|
+
if (tagName === 'dl') {
|
|
474
|
+
return {
|
|
475
|
+
type: 'definitionList',
|
|
476
|
+
children: parseChildren(node, newFormatting, listContext)
|
|
477
|
+
};
|
|
478
|
+
}
|
|
479
|
+
if (tagName === 'dt') {
|
|
480
|
+
return {
|
|
481
|
+
type: 'definitionTerm',
|
|
482
|
+
children: parseChildren(node, newFormatting, listContext)
|
|
483
|
+
};
|
|
484
|
+
}
|
|
485
|
+
if (tagName === 'dd') {
|
|
486
|
+
return {
|
|
487
|
+
type: 'definitionDescription',
|
|
488
|
+
children: parseChildren(node, newFormatting, listContext)
|
|
489
|
+
};
|
|
490
|
+
}
|
|
491
|
+
if (tagName === 'abbr') {
|
|
492
|
+
const title = node.attributes?.title;
|
|
493
|
+
const children = parseChildren(node, newFormatting, listContext);
|
|
494
|
+
if (title) {
|
|
495
|
+
children.forEach(c => {
|
|
496
|
+
if (c.type === 'text') {
|
|
497
|
+
c.metadata = { ...c.metadata, abbreviationTitle: title };
|
|
498
|
+
}
|
|
499
|
+
});
|
|
500
|
+
}
|
|
501
|
+
return children;
|
|
502
|
+
}
|
|
503
|
+
if (tagName === 'cite' && node.attributes?.['data-citation-key'] !== undefined) {
|
|
504
|
+
const citationKey = node.attributes['data-citation-key'];
|
|
505
|
+
return {
|
|
506
|
+
type: 'text',
|
|
507
|
+
text: citationKey,
|
|
508
|
+
formatting: Object.keys(newFormatting).length > 0 ? { ...newFormatting } : undefined,
|
|
509
|
+
metadata: { citationKey }
|
|
510
|
+
};
|
|
511
|
+
}
|
|
333
512
|
if (tagName === 'ul' || tagName === 'ol') {
|
|
334
513
|
const isNewTopLevel = !listContext;
|
|
335
514
|
const newListContext = {
|
|
336
515
|
listId: isNewTopLevel ? `html-list-${htmlListIdCounter++}` : listContext.listId,
|
|
337
516
|
type: tagName === 'ol' ? 'ordered' : 'unordered',
|
|
338
517
|
level: isNewTopLevel ? 0 : listContext.level + 1,
|
|
339
|
-
counters: isNewTopLevel ? {} : { ...listContext.counters } // Clone to avoid side effects on parent levels
|
|
518
|
+
counters: isNewTopLevel ? {} : { ...listContext.counters }, // Clone to avoid side effects on parent levels
|
|
519
|
+
isTask: node.attributes?.['data-type'] === 'taskList'
|
|
340
520
|
};
|
|
341
521
|
// Initialize counter for this level
|
|
342
522
|
if (tagName === 'ol' && node.attributes?.start) {
|
|
@@ -362,6 +542,13 @@ const parseHtml = async (buffer, config) => {
|
|
|
362
542
|
const children = parseChildren(node, newFormatting, listContext);
|
|
363
543
|
const nestedLists = children.filter(c => c.type === 'list');
|
|
364
544
|
const selfChildren = children.filter(c => c.type !== 'list');
|
|
545
|
+
let isTask;
|
|
546
|
+
let checked;
|
|
547
|
+
if (listContext?.isTask) {
|
|
548
|
+
isTask = true;
|
|
549
|
+
const dataChecked = node.attributes?.['data-checked'];
|
|
550
|
+
checked = dataChecked !== undefined ? dataChecked === 'true' : (findNestedCheckboxChecked(node) ?? false);
|
|
551
|
+
}
|
|
365
552
|
const selfNode = {
|
|
366
553
|
type: 'list',
|
|
367
554
|
text: selfChildren.map(c => c.text || '').join(''),
|
|
@@ -371,16 +558,21 @@ const parseHtml = async (buffer, config) => {
|
|
|
371
558
|
alignment: newFormatting.alignment || 'left',
|
|
372
559
|
listId: listContext?.listId || 'html-list-none',
|
|
373
560
|
itemIndex: (listContext?.counters[listContext.level] ?? 1) - 1,
|
|
374
|
-
anchorIds: anchorIds.length > 0 ? anchorIds : undefined
|
|
561
|
+
anchorIds: anchorIds.length > 0 ? anchorIds : undefined,
|
|
562
|
+
isTask,
|
|
563
|
+
checked
|
|
375
564
|
},
|
|
376
565
|
children: selfChildren
|
|
377
566
|
};
|
|
378
567
|
return [selfNode, ...nestedLists];
|
|
379
568
|
}
|
|
380
569
|
if (tagName === 'table') {
|
|
570
|
+
// CustomTable (inscript-editor) renders data-align on the <table> itself.
|
|
571
|
+
const tableAlignAttr = node.attributes?.['data-align'];
|
|
572
|
+
const tableAlign = ['left', 'center', 'right'].includes(tableAlignAttr) ? tableAlignAttr : undefined;
|
|
381
573
|
const tableNode = {
|
|
382
574
|
type: 'table',
|
|
383
|
-
metadata: { anchorIds: anchorIds.length > 0 ? anchorIds : undefined },
|
|
575
|
+
metadata: { anchorIds: anchorIds.length > 0 ? anchorIds : undefined, align: tableAlign },
|
|
384
576
|
children: parseChildren(node, newFormatting, listContext)
|
|
385
577
|
};
|
|
386
578
|
if (config.includeRawContent) {
|
|
@@ -399,8 +591,18 @@ const parseHtml = async (buffer, config) => {
|
|
|
399
591
|
return rowNode;
|
|
400
592
|
}
|
|
401
593
|
if (tagName === 'td' || tagName === 'th') {
|
|
594
|
+
// Merged cells: mirrors the colspan/rowspan reading already done in
|
|
595
|
+
// MarkdownParser's inline HTML-table handler.
|
|
596
|
+
const colSpanAttr = node.attributes?.colspan;
|
|
597
|
+
const rowSpanAttr = node.attributes?.rowspan;
|
|
598
|
+
const colSpan = colSpanAttr ? parseInt(colSpanAttr, 10) : undefined;
|
|
599
|
+
const rowSpan = rowSpanAttr ? parseInt(rowSpanAttr, 10) : undefined;
|
|
402
600
|
const cellNode = {
|
|
403
601
|
type: 'cell',
|
|
602
|
+
metadata: {
|
|
603
|
+
colSpan: colSpan && !isNaN(colSpan) ? colSpan : undefined,
|
|
604
|
+
rowSpan: rowSpan && !isNaN(rowSpan) ? rowSpan : undefined
|
|
605
|
+
},
|
|
404
606
|
children: parseChildren(node, newFormatting, listContext)
|
|
405
607
|
};
|
|
406
608
|
if (config.includeRawContent) {
|
|
@@ -411,6 +613,14 @@ const parseHtml = async (buffer, config) => {
|
|
|
411
613
|
if (tagName === 'img') {
|
|
412
614
|
const src = node.attributes?.src;
|
|
413
615
|
const alt = node.attributes?.alt;
|
|
616
|
+
// CustomImage (inscript-editor) renders data-width/data-align, falling back to
|
|
617
|
+
// parsing the inline style for consumers that only emit the CSS.
|
|
618
|
+
const imgStyle = node.attributes?.style || '';
|
|
619
|
+
const width = node.attributes?.['data-width'] || imgStyle.match(/width:\s*([^;]+)/)?.[1]?.trim();
|
|
620
|
+
const alignAttr = node.attributes?.['data-align']
|
|
621
|
+
|| (imgStyle.includes('margin-left: 0') && !imgStyle.includes('margin-right: 0') ? 'left'
|
|
622
|
+
: (imgStyle.includes('margin-right: 0') && !imgStyle.includes('margin-left: 0') ? 'right' : undefined));
|
|
623
|
+
const align = ['left', 'center', 'right'].includes(alignAttr) ? alignAttr : undefined;
|
|
414
624
|
let imageNode;
|
|
415
625
|
if (src?.startsWith('data:')) {
|
|
416
626
|
const match = src.match(/^data:([^;]+);base64,(.*)$/);
|
|
@@ -429,7 +639,9 @@ const parseHtml = async (buffer, config) => {
|
|
|
429
639
|
type: 'image',
|
|
430
640
|
metadata: {
|
|
431
641
|
attachmentName: name,
|
|
432
|
-
altText: alt
|
|
642
|
+
altText: alt,
|
|
643
|
+
width,
|
|
644
|
+
align
|
|
433
645
|
}
|
|
434
646
|
};
|
|
435
647
|
}
|
|
@@ -438,7 +650,9 @@ const parseHtml = async (buffer, config) => {
|
|
|
438
650
|
type: 'image',
|
|
439
651
|
metadata: {
|
|
440
652
|
url: src,
|
|
441
|
-
altText: alt
|
|
653
|
+
altText: alt,
|
|
654
|
+
width,
|
|
655
|
+
align
|
|
442
656
|
}
|
|
443
657
|
};
|
|
444
658
|
}
|
|
@@ -449,7 +663,9 @@ const parseHtml = async (buffer, config) => {
|
|
|
449
663
|
metadata: {
|
|
450
664
|
url: src,
|
|
451
665
|
altText: alt,
|
|
452
|
-
anchorIds: anchorIds.length > 0 ? anchorIds : undefined
|
|
666
|
+
anchorIds: anchorIds.length > 0 ? anchorIds : undefined,
|
|
667
|
+
width,
|
|
668
|
+
align
|
|
453
669
|
}
|
|
454
670
|
};
|
|
455
671
|
}
|
|
@@ -460,8 +676,16 @@ const parseHtml = async (buffer, config) => {
|
|
|
460
676
|
}
|
|
461
677
|
if (tagName === 'a') {
|
|
462
678
|
const href = node.attributes?.href;
|
|
679
|
+
const wikilinkPage = node.attributes?.['data-wikilink-page'];
|
|
463
680
|
const children = parseChildren(node, newFormatting, listContext);
|
|
464
|
-
if (
|
|
681
|
+
if (wikilinkPage !== undefined) {
|
|
682
|
+
children.forEach(c => {
|
|
683
|
+
if (c.type === 'text') {
|
|
684
|
+
c.metadata = { ...c.metadata, link: wikilinkPage, linkType: 'internal', wikilink: true };
|
|
685
|
+
}
|
|
686
|
+
});
|
|
687
|
+
}
|
|
688
|
+
else if (href) {
|
|
465
689
|
const linkType = href.startsWith('#') ? 'internal' : 'external';
|
|
466
690
|
children.forEach(c => {
|
|
467
691
|
if (c.type === 'text') {
|
|
@@ -509,6 +733,42 @@ const parseHtml = async (buffer, config) => {
|
|
|
509
733
|
}
|
|
510
734
|
return null;
|
|
511
735
|
};
|
|
736
|
+
// Extract <section data-footnotes> up front so its definitions are available to
|
|
737
|
+
// <sup data-footnote-ref> references encountered anywhere earlier in the body.
|
|
738
|
+
const findFootnotesSection = (n) => {
|
|
739
|
+
if (n.tagName === 'section' && n.attributes?.['data-footnotes'] !== undefined)
|
|
740
|
+
return n;
|
|
741
|
+
for (const child of n.children) {
|
|
742
|
+
const found = findFootnotesSection(child);
|
|
743
|
+
if (found)
|
|
744
|
+
return found;
|
|
745
|
+
}
|
|
746
|
+
return undefined;
|
|
747
|
+
};
|
|
748
|
+
const footnotesSectionNode = findFootnotesSection(body);
|
|
749
|
+
if (footnotesSectionNode) {
|
|
750
|
+
for (const item of footnotesSectionNode.children) {
|
|
751
|
+
if (item.type !== 'element')
|
|
752
|
+
continue;
|
|
753
|
+
const key = item.attributes?.['data-footnote-id'];
|
|
754
|
+
if (!key)
|
|
755
|
+
continue;
|
|
756
|
+
// Strip the generated back-reference link ("↩") - it's round-trip plumbing,
|
|
757
|
+
// not part of the footnote's actual content.
|
|
758
|
+
const filteredChildren = item.children.filter(c => !(c.tagName === 'a' && (c.attributes?.href || '').startsWith('#footnote-ref-')));
|
|
759
|
+
const contentNodes = [];
|
|
760
|
+
for (const child of filteredChildren) {
|
|
761
|
+
const parsed = parseNode(child);
|
|
762
|
+
if (parsed) {
|
|
763
|
+
if (Array.isArray(parsed))
|
|
764
|
+
contentNodes.push(...parsed);
|
|
765
|
+
else
|
|
766
|
+
contentNodes.push(parsed);
|
|
767
|
+
}
|
|
768
|
+
}
|
|
769
|
+
footnoteDefinitions.set(key, contentNodes);
|
|
770
|
+
}
|
|
771
|
+
}
|
|
512
772
|
for (const child of body.children) {
|
|
513
773
|
const parsed = parseNode(child);
|
|
514
774
|
if (parsed) {
|
|
@@ -539,8 +799,12 @@ const parseHtml = async (buffer, config) => {
|
|
|
539
799
|
return node.text || '';
|
|
540
800
|
if (node.type === 'break')
|
|
541
801
|
return '\n';
|
|
802
|
+
// Childless nodes still carry meaningful text - fall back to it instead of
|
|
803
|
+
// silently vanishing from plain-text/RAG-chunk output.
|
|
804
|
+
if (node.type === 'embed')
|
|
805
|
+
return node.metadata?.url || '';
|
|
542
806
|
if (node.children) {
|
|
543
|
-
const isBlock = ['table', 'row', 'list', 'sheet', 'slide'].includes(node.type);
|
|
807
|
+
const isBlock = ['table', 'row', 'list', 'sheet', 'slide', 'admonition', 'definitionList'].includes(node.type);
|
|
544
808
|
return node.children.map(getText).join(isBlock ? config.newlineDelimiter : '');
|
|
545
809
|
}
|
|
546
810
|
return '';
|