officeparser 7.2.3 → 7.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +277 -17
- package/dist/OfficeConverter.d.ts +1 -1
- package/dist/OfficeConverter.js +3 -0
- package/dist/OfficeGenerator.js +4 -0
- package/dist/OfficeParser.d.ts +2 -0
- package/dist/OfficeParser.js +6 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +5 -2
- package/dist/defaults.js +12 -0
- package/dist/generators/BaseGenerator.d.ts +34 -1
- package/dist/generators/BaseGenerator.js +98 -0
- package/dist/generators/CsvGenerator.d.ts +9 -1
- package/dist/generators/CsvGenerator.js +28 -16
- package/dist/generators/EpubGenerator.d.ts +43 -0
- package/dist/generators/EpubGenerator.js +312 -0
- package/dist/generators/HtmlGenerator.d.ts +12 -0
- package/dist/generators/HtmlGenerator.js +378 -61
- package/dist/generators/MarkdownGenerator.d.ts +28 -5
- package/dist/generators/MarkdownGenerator.js +432 -51
- package/dist/generators/PdfGenerator.js +32 -0
- package/dist/generators/RtfGenerator.js +47 -22
- package/dist/generators/TextGenerator.js +98 -11
- package/dist/index.d.ts +1 -0
- package/dist/index.js +1 -0
- package/dist/officeparser.browser.d.ts +427 -20
- package/dist/officeparser.browser.iife.js +338 -206
- package/dist/officeparser.browser.mjs +346 -214
- package/dist/officeparser.browser.slim.d.ts +427 -20
- package/dist/officeparser.browser.slim.iife.js +346 -214
- package/dist/officeparser.browser.slim.mjs +346 -214
- package/dist/parsers/EpubParser.d.ts +8 -0
- package/dist/parsers/EpubParser.js +217 -0
- package/dist/parsers/ExcelParser.js +2 -0
- package/dist/parsers/HtmlParser.js +507 -48
- package/dist/parsers/MarkdownParser.js +704 -92
- package/dist/parsers/OpenOfficeParser.js +128 -20
- package/dist/parsers/PdfParser.js +4 -1
- package/dist/parsers/PowerPointParser.js +1 -0
- package/dist/parsers/WordParser.js +1 -0
- package/dist/sbom.cdx.json +1695 -0
- package/dist/types.d.ts +427 -20
- package/dist/types.js +8 -0
- package/dist/utils/configUtils.js +53 -4
- package/dist/utils/errorUtils.js +7 -3
- package/dist/utils/sanitize.d.ts +139 -0
- package/dist/utils/sanitize.js +318 -0
- package/dist/utils/xmlUtils.js +2 -2
- package/dist/utils/zipUtils.js +76 -26
- package/package.json +16 -12
|
@@ -1,11 +1,25 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
3
|
exports.parseHtml = void 0;
|
|
4
|
+
const types_js_1 = require("../types.js");
|
|
4
5
|
const astUtils_js_1 = require("../utils/astUtils.js");
|
|
5
6
|
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
7
|
+
const sanitize_js_1 = require("../utils/sanitize.js");
|
|
8
|
+
/**
|
|
9
|
+
* Maximum element nesting depth accepted from an HTML/XHTML source before the parser gives up
|
|
10
|
+
* with a typed error rather than letting the recursion overflow the call stack. See the guard in
|
|
11
|
+
* `parseNode` for why this value and not a larger one.
|
|
12
|
+
*/
|
|
13
|
+
const MAX_HTML_NESTING_DEPTH = 256;
|
|
6
14
|
const parseAttributes = (attrString) => {
|
|
7
15
|
const attrs = {};
|
|
8
|
-
|
|
16
|
+
// Attribute names follow the HTML5 rule - any character except whitespace and
|
|
17
|
+
// " ' > / = - rather than a hand-picked allowlist. The previous class
|
|
18
|
+
// ([a-zA-Z0-9\-:]) silently split a legal name on any character outside it, so
|
|
19
|
+
// `data_foo="x"` produced TWO attributes: `data` (empty) and an invented
|
|
20
|
+
// `foo="x"` that was never in the source. Harmless while nothing read unknown
|
|
21
|
+
// attributes; not harmless once they can be replayed into generated output.
|
|
22
|
+
const regex = /([^\s"'>/=]+)(?:\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s>]+)))?/g;
|
|
9
23
|
let match;
|
|
10
24
|
while ((match = regex.exec(attrString)) !== null) {
|
|
11
25
|
const name = match[1].toLowerCase();
|
|
@@ -14,6 +28,101 @@ const parseAttributes = (attrString) => {
|
|
|
14
28
|
}
|
|
15
29
|
return attrs;
|
|
16
30
|
};
|
|
31
|
+
/**
|
|
32
|
+
* Splits an inline `style` attribute into a property -> value map.
|
|
33
|
+
*
|
|
34
|
+
* Replaces substring matching (`styleAttr.includes('font-weight: bold')`) and unanchored regexes
|
|
35
|
+
* (`/color:\s*([^;]+)/`), which were wrong in both directions:
|
|
36
|
+
* - false positives: `color:` matched inside `background-color:`, so
|
|
37
|
+
* `"background-color: red; color: blue"` yielded color=red - the *wrong* value, not merely a
|
|
38
|
+
* spurious one - and `width:` matched inside `max-width:`, so the ubiquitous responsive-image
|
|
39
|
+
* style `max-width: 100%` was read as an explicit width.
|
|
40
|
+
* - false negatives: `font-weight:bold` without a space, `font-weight: 700`, and `bolder` were
|
|
41
|
+
* all missed, as was `line-through` inside `text-decoration: underline line-through`.
|
|
42
|
+
*
|
|
43
|
+
* Splitting is quote- and paren-aware so a semicolon inside `url(data:image/png;base64,...)` or a
|
|
44
|
+
* quoted font stack doesn't shatter the declaration. `!important` is stripped from values, since
|
|
45
|
+
* substring matching used to tolerate it and exact comparison otherwise would not - dropping it
|
|
46
|
+
* would be a silent regression rather than the intended fix.
|
|
47
|
+
*/
|
|
48
|
+
const parseStyleDeclarations = (styleAttr) => {
|
|
49
|
+
const decls = new Map();
|
|
50
|
+
if (!styleAttr)
|
|
51
|
+
return decls;
|
|
52
|
+
let depth = 0;
|
|
53
|
+
let quote = null;
|
|
54
|
+
let current = '';
|
|
55
|
+
const chunks = [];
|
|
56
|
+
for (const ch of styleAttr) {
|
|
57
|
+
if (quote) {
|
|
58
|
+
if (ch === quote)
|
|
59
|
+
quote = null;
|
|
60
|
+
}
|
|
61
|
+
else if (ch === '"' || ch === '\'') {
|
|
62
|
+
quote = ch;
|
|
63
|
+
}
|
|
64
|
+
else if (ch === '(') {
|
|
65
|
+
depth++;
|
|
66
|
+
}
|
|
67
|
+
else if (ch === ')') {
|
|
68
|
+
if (depth > 0)
|
|
69
|
+
depth--;
|
|
70
|
+
}
|
|
71
|
+
else if (ch === ';' && depth === 0) {
|
|
72
|
+
chunks.push(current);
|
|
73
|
+
current = '';
|
|
74
|
+
continue;
|
|
75
|
+
}
|
|
76
|
+
current += ch;
|
|
77
|
+
}
|
|
78
|
+
chunks.push(current);
|
|
79
|
+
for (const chunk of chunks) {
|
|
80
|
+
const idx = chunk.indexOf(':');
|
|
81
|
+
if (idx === -1)
|
|
82
|
+
continue;
|
|
83
|
+
const prop = chunk.slice(0, idx).trim().toLowerCase();
|
|
84
|
+
if (!prop)
|
|
85
|
+
continue;
|
|
86
|
+
const value = chunk.slice(idx + 1).trim().replace(/\s*!\s*important\s*$/i, '').trim();
|
|
87
|
+
if (value)
|
|
88
|
+
decls.set(prop, value);
|
|
89
|
+
}
|
|
90
|
+
return decls;
|
|
91
|
+
};
|
|
92
|
+
/**
|
|
93
|
+
* Reads a declaration, also accepting the `-webkit-`/`-moz-`/`-ms-`/`-o-` prefixed spelling so a
|
|
94
|
+
* vendor-prefixed property keeps matching (substring matching used to catch those by accident).
|
|
95
|
+
*/
|
|
96
|
+
const getDeclaration = (decls, prop) => decls.get(prop)
|
|
97
|
+
?? decls.get(`-webkit-${prop}`)
|
|
98
|
+
?? decls.get(`-moz-${prop}`)
|
|
99
|
+
?? decls.get(`-ms-${prop}`)
|
|
100
|
+
?? decls.get(`-o-${prop}`);
|
|
101
|
+
/**
|
|
102
|
+
* Returns the first family from a `font-family` stack, respecting quotes so a quoted family name
|
|
103
|
+
* containing a comma (`'Fira, A', serif`) isn't split through the middle of its own name.
|
|
104
|
+
*/
|
|
105
|
+
const firstFontFamily = (fontFamily) => {
|
|
106
|
+
let quote = null;
|
|
107
|
+
let first = '';
|
|
108
|
+
for (const ch of fontFamily) {
|
|
109
|
+
if (quote) {
|
|
110
|
+
if (ch === quote) {
|
|
111
|
+
quote = null;
|
|
112
|
+
continue;
|
|
113
|
+
}
|
|
114
|
+
}
|
|
115
|
+
else if (ch === '"' || ch === '\'') {
|
|
116
|
+
quote = ch;
|
|
117
|
+
continue;
|
|
118
|
+
}
|
|
119
|
+
else if (ch === ',') {
|
|
120
|
+
break;
|
|
121
|
+
}
|
|
122
|
+
first += ch;
|
|
123
|
+
}
|
|
124
|
+
return first.trim();
|
|
125
|
+
};
|
|
17
126
|
const parseHtmlTree = (html) => {
|
|
18
127
|
const root = { type: 'element', tagName: 'root', children: [], attributes: {} };
|
|
19
128
|
let current = root;
|
|
@@ -36,14 +145,16 @@ const parseHtmlTree = (html) => {
|
|
|
36
145
|
cursor = commentEnd !== -1 ? commentEnd + 3 : html.length;
|
|
37
146
|
continue;
|
|
38
147
|
}
|
|
39
|
-
|
|
40
|
-
|
|
148
|
+
// indexOf (not substring().match) so scanning for the tag end is O(1) in
|
|
149
|
+
// allocation — a document with many "<" chars would otherwise be O(n^2).
|
|
150
|
+
const tagEndIdx = html.indexOf('>', tagStart);
|
|
151
|
+
if (tagEndIdx === -1) {
|
|
41
152
|
const text = html.substring(tagStart);
|
|
42
153
|
current.children.push({ type: 'text', text, children: [], parent: current });
|
|
43
154
|
break;
|
|
44
155
|
}
|
|
45
|
-
const tagContent = html.substring(tagStart + 1,
|
|
46
|
-
cursor =
|
|
156
|
+
const tagContent = html.substring(tagStart + 1, tagEndIdx);
|
|
157
|
+
cursor = tagEndIdx + 1;
|
|
47
158
|
const isClosing = tagContent.startsWith('/');
|
|
48
159
|
const isSelfClosing = tagContent.endsWith('/');
|
|
49
160
|
const tagCore = tagContent.replace(/^\/|\/$/g, '').trim();
|
|
@@ -77,16 +188,20 @@ const parseHtmlTree = (html) => {
|
|
|
77
188
|
if (!isSelfClosing && !voidElements.has(tagName)) {
|
|
78
189
|
current = node;
|
|
79
190
|
if (tagName === 'script' || tagName === 'style') {
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
191
|
+
// Case-insensitive search from `cursor` via a sticky-ish regex, instead of
|
|
192
|
+
// lower-casing the whole document on every <script>/<style> (was O(n^2)).
|
|
193
|
+
// tagName is validated to /^[a-z0-9-]+$/ above, so it's safe to interpolate.
|
|
194
|
+
const closeRe = new RegExp(`</${tagName}>`, 'gi');
|
|
195
|
+
closeRe.lastIndex = cursor;
|
|
196
|
+
const closeMatch = closeRe.exec(html);
|
|
197
|
+
if (closeMatch) {
|
|
83
198
|
node.children.push({
|
|
84
199
|
type: 'text',
|
|
85
|
-
text: html.substring(cursor,
|
|
200
|
+
text: html.substring(cursor, closeMatch.index),
|
|
86
201
|
children: [],
|
|
87
202
|
parent: node
|
|
88
203
|
});
|
|
89
|
-
cursor =
|
|
204
|
+
cursor = closeMatch.index + closeMatch[0].length;
|
|
90
205
|
current = node.parent;
|
|
91
206
|
}
|
|
92
207
|
}
|
|
@@ -183,7 +298,82 @@ const parseHtml = async (buffer, config) => {
|
|
|
183
298
|
}
|
|
184
299
|
const content = [];
|
|
185
300
|
let htmlListIdCounter = 1;
|
|
186
|
-
|
|
301
|
+
// Finds the checked state from a nested <input type="checkbox"> (GFM task-list items
|
|
302
|
+
// nest it inside a <label>, so it isn't a direct child of the <li>).
|
|
303
|
+
const findNestedCheckboxChecked = (n) => {
|
|
304
|
+
if (n.tagName === 'input' && (n.attributes?.type || '').toLowerCase() === 'checkbox') {
|
|
305
|
+
return 'checked' in (n.attributes || {});
|
|
306
|
+
}
|
|
307
|
+
for (const child of n.children) {
|
|
308
|
+
const found = findNestedCheckboxChecked(child);
|
|
309
|
+
if (found !== undefined)
|
|
310
|
+
return found;
|
|
311
|
+
}
|
|
312
|
+
return undefined;
|
|
313
|
+
};
|
|
314
|
+
// Populated from a <section data-footnotes> block (found and parsed before the main
|
|
315
|
+
// body loop, since references can appear anywhere earlier in the document) and
|
|
316
|
+
// consulted by parseChildren's <sup data-footnote-ref> handling below.
|
|
317
|
+
const footnoteDefinitions = new Map();
|
|
318
|
+
// --- Generic attribute pass-through (htmlParserConfig.preserveAttributes) ---------------
|
|
319
|
+
// Captures attributes no typed metadata field consumed, so they can be replayed on
|
|
320
|
+
// generation. Everything here is a *defence-in-depth* filter: HtmlGenerator sanitizes again
|
|
321
|
+
// on the way out, because an AST can be built programmatically rather than parsed.
|
|
322
|
+
const preserveAttributes = config.htmlParserConfig?.preserveAttributes === true;
|
|
323
|
+
// `style` is already consumed wholesale into TextFormatting/metadata above, and `id` is
|
|
324
|
+
// consumed into anchorIds and re-emitted by the generator - carrying either would duplicate
|
|
325
|
+
// an attribute the generator composes itself. `class` is deliberately NOT excluded: the
|
|
326
|
+
// generator's class attribute is built purely from style-mapping and never from a parsed
|
|
327
|
+
// `class`, so without this a plain `<p class="lead">` loses "lead" entirely. The generator
|
|
328
|
+
// merges it into that attribute rather than emitting a second one.
|
|
329
|
+
const GENERATOR_OWNED_ATTRS = new Set(['id', 'style']);
|
|
330
|
+
/**
|
|
331
|
+
* Captures the attributes of `node` that `consumed` didn't claim.
|
|
332
|
+
* Returns undefined when nothing survives, so the field stays absent rather than `{}`.
|
|
333
|
+
*/
|
|
334
|
+
const collectHtmlAttributes = (node, consumed) => {
|
|
335
|
+
if (!preserveAttributes || !node.attributes)
|
|
336
|
+
return undefined;
|
|
337
|
+
const consumedSet = new Set(consumed.map(c => c.toLowerCase()));
|
|
338
|
+
const bag = {};
|
|
339
|
+
for (const [rawKey, value] of Object.entries(node.attributes)) {
|
|
340
|
+
const key = rawKey.toLowerCase();
|
|
341
|
+
if (consumedSet.has(key) || GENERATOR_OWNED_ATTRS.has(key))
|
|
342
|
+
continue;
|
|
343
|
+
// Event handlers are never carried, at any layer, with no opt-in.
|
|
344
|
+
if (/^on/i.test(key))
|
|
345
|
+
continue;
|
|
346
|
+
// srcdoc holds a whole HTML document; it cannot be safely escaped into an attribute.
|
|
347
|
+
if (key === 'srcdoc')
|
|
348
|
+
continue;
|
|
349
|
+
// Reject anything that isn't a plain attribute name outright - a key containing a
|
|
350
|
+
// quote or '=' is the shape an attribute-injection payload takes.
|
|
351
|
+
if (!(0, sanitize_js_1.isSafeHtmlAttributeName)(key))
|
|
352
|
+
continue;
|
|
353
|
+
bag[key] = value;
|
|
354
|
+
}
|
|
355
|
+
return Object.keys(bag).length > 0 ? bag : undefined;
|
|
356
|
+
};
|
|
357
|
+
const parseNode = (node, currentFormatting = {}, listContext, depth = 0) => {
|
|
358
|
+
// Guard against a maliciously deep element tree (e.g. tens of thousands of nested
|
|
359
|
+
// <div>) recursing until the call stack overflows.
|
|
360
|
+
//
|
|
361
|
+
// The previous limit of 1000 could never fire: measured overflow is around 800 and
|
|
362
|
+
// varies run to run (796/862/796 on three identical runs), so the RangeError always
|
|
363
|
+
// arrived first and the typed error this guard exists to produce never did. Failure was
|
|
364
|
+
// still graceful - it surfaces as a wrapped Error, not a crash - which is why this was a
|
|
365
|
+
// dead guard rather than a denial of service.
|
|
366
|
+
//
|
|
367
|
+
// 256 is chosen to hold across engines rather than tuned to one. It is far below the
|
|
368
|
+
// lowest overflow observed here and leaves room for a smaller frame budget on older V8
|
|
369
|
+
// (the supported floor is Node 18), while sitting orders of magnitude above real
|
|
370
|
+
// content: the bundled HTML and EPUB fixtures reach an AST depth of 8.
|
|
371
|
+
// Per node, alongside the depth guard: the two together are what make a hostile
|
|
372
|
+
// document both bounded and cancellable rather than only bounded.
|
|
373
|
+
(0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
|
|
374
|
+
if (depth > MAX_HTML_NESTING_DEPTH) {
|
|
375
|
+
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.MAX_NESTING_DEPTH_EXCEEDED);
|
|
376
|
+
}
|
|
187
377
|
if (node.type === 'text') {
|
|
188
378
|
let decodedText = (node.text || '')
|
|
189
379
|
.replace(/ /g, ' ')
|
|
@@ -229,29 +419,43 @@ const parseHtml = async (buffer, config) => {
|
|
|
229
419
|
const styleAttr = node.attributes?.style || '';
|
|
230
420
|
const alignAttr = node.attributes?.align || '';
|
|
231
421
|
if (styleAttr || alignAttr) {
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
422
|
+
const decls = parseStyleDeclarations(styleAttr);
|
|
423
|
+
// `bold`, `bolder`, and any weight >= 600 are all bold; the old substring check
|
|
424
|
+
// only ever saw the literal "font-weight: bold".
|
|
425
|
+
const weight = getDeclaration(decls, 'font-weight');
|
|
426
|
+
if (weight) {
|
|
427
|
+
const numericWeight = parseInt(weight, 10);
|
|
428
|
+
if (weight === 'bold' || weight === 'bolder' || (!isNaN(numericWeight) && numericWeight >= 600)) {
|
|
429
|
+
newFormatting.bold = true;
|
|
430
|
+
}
|
|
431
|
+
}
|
|
432
|
+
if (getDeclaration(decls, 'font-style') === 'italic')
|
|
235
433
|
newFormatting.italic = true;
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
const
|
|
247
|
-
if (
|
|
248
|
-
newFormatting.
|
|
249
|
-
const
|
|
250
|
-
if (
|
|
251
|
-
newFormatting.
|
|
252
|
-
const
|
|
253
|
-
if (
|
|
254
|
-
newFormatting.
|
|
434
|
+
// text-decoration is a shorthand that can carry several keywords at once, so
|
|
435
|
+
// "underline line-through" has to set both flags rather than only the first.
|
|
436
|
+
const decoration = getDeclaration(decls, 'text-decoration') ?? getDeclaration(decls, 'text-decoration-line');
|
|
437
|
+
if (decoration) {
|
|
438
|
+
const parts = decoration.split(/\s+/);
|
|
439
|
+
if (parts.includes('underline'))
|
|
440
|
+
newFormatting.underline = true;
|
|
441
|
+
if (parts.includes('line-through'))
|
|
442
|
+
newFormatting.strikethrough = true;
|
|
443
|
+
}
|
|
444
|
+
const color = decls.get('color');
|
|
445
|
+
if (color)
|
|
446
|
+
newFormatting.color = color;
|
|
447
|
+
const background = getDeclaration(decls, 'background-color');
|
|
448
|
+
if (background)
|
|
449
|
+
newFormatting.backgroundColor = background;
|
|
450
|
+
const size = getDeclaration(decls, 'font-size');
|
|
451
|
+
if (size)
|
|
452
|
+
newFormatting.size = size;
|
|
453
|
+
const fontFamily = getDeclaration(decls, 'font-family');
|
|
454
|
+
if (fontFamily)
|
|
455
|
+
newFormatting.font = firstFontFamily(fontFamily);
|
|
456
|
+
const textAlign = getDeclaration(decls, 'text-align')?.toLowerCase();
|
|
457
|
+
if (textAlign && ['left', 'center', 'right', 'justify'].includes(textAlign)) {
|
|
458
|
+
newFormatting.alignment = textAlign;
|
|
255
459
|
}
|
|
256
460
|
else if (alignAttr) {
|
|
257
461
|
const align = alignAttr.toLowerCase();
|
|
@@ -264,7 +468,29 @@ const parseHtml = async (buffer, config) => {
|
|
|
264
468
|
const parseChildren = (n, fmt, lCtx) => {
|
|
265
469
|
const kids = [];
|
|
266
470
|
for (const child of n.children) {
|
|
267
|
-
|
|
471
|
+
// Footnote/endnote reference: attach as .notes on the preceding node
|
|
472
|
+
// instead of inserting a visible node, matching WordParser's convention.
|
|
473
|
+
if (child.type === 'element' && child.tagName === 'sup' && child.attributes?.['data-footnote-ref'] !== undefined) {
|
|
474
|
+
const key = child.attributes['data-footnote-ref'];
|
|
475
|
+
const definition = footnoteDefinitions.get(key);
|
|
476
|
+
const noteNode = {
|
|
477
|
+
type: 'note',
|
|
478
|
+
text: (definition || []).map(d => d.text || '').join(''),
|
|
479
|
+
children: definition || [],
|
|
480
|
+
metadata: { noteType: 'footnote', noteId: key }
|
|
481
|
+
};
|
|
482
|
+
if (kids.length > 0) {
|
|
483
|
+
const target = kids[kids.length - 1];
|
|
484
|
+
if (!target.notes)
|
|
485
|
+
target.notes = [];
|
|
486
|
+
target.notes.push(noteNode);
|
|
487
|
+
}
|
|
488
|
+
else {
|
|
489
|
+
kids.push({ type: 'text', text: '', notes: [noteNode] });
|
|
490
|
+
}
|
|
491
|
+
continue;
|
|
492
|
+
}
|
|
493
|
+
const parsed = parseNode(child, fmt, lCtx, depth + 1);
|
|
268
494
|
if (parsed) {
|
|
269
495
|
if (Array.isArray(parsed))
|
|
270
496
|
kids.push(...parsed);
|
|
@@ -274,6 +500,94 @@ const parseHtml = async (buffer, config) => {
|
|
|
274
500
|
}
|
|
275
501
|
return kids;
|
|
276
502
|
};
|
|
503
|
+
// YouTube embeds: inscript-editor's Youtube node renders
|
|
504
|
+
// <div data-youtube-video="ID" data-width="…" data-align="…">…<iframe…></div>.
|
|
505
|
+
// Recognise both the wrapper div and a bare iframe so externally-authored HTML
|
|
506
|
+
// (and a saved-then-reopened .md that fell back to raw HTML) both round-trip.
|
|
507
|
+
if (tagName === 'div' && node.attributes?.['data-youtube-video'] !== undefined) {
|
|
508
|
+
const videoId = node.attributes['data-youtube-video'] || '';
|
|
509
|
+
const width = node.attributes?.['data-width'];
|
|
510
|
+
const embedAlignAttr = node.attributes?.['data-align'];
|
|
511
|
+
const embedAlign = ['left', 'center', 'right'].includes(embedAlignAttr) ? embedAlignAttr : undefined;
|
|
512
|
+
const embedUrl = videoId ? `https://www.youtube.com/watch?v=${videoId}` : undefined;
|
|
513
|
+
const embedNode = {
|
|
514
|
+
type: 'embed',
|
|
515
|
+
// Childless nodes need .text so generic AST consumers (toText, chunking)
|
|
516
|
+
// don't silently drop them.
|
|
517
|
+
text: embedUrl,
|
|
518
|
+
metadata: {
|
|
519
|
+
embedType: 'youtube',
|
|
520
|
+
videoId,
|
|
521
|
+
url: embedUrl,
|
|
522
|
+
width,
|
|
523
|
+
align: embedAlign
|
|
524
|
+
}
|
|
525
|
+
};
|
|
526
|
+
if (config.includeRawContent)
|
|
527
|
+
embedNode.rawContent = '<div data-youtube-video>...</div>';
|
|
528
|
+
return embedNode;
|
|
529
|
+
}
|
|
530
|
+
if (tagName === 'iframe') {
|
|
531
|
+
const src = node.attributes?.src || '';
|
|
532
|
+
const ytMatch = /youtube(?:-nocookie)?\.com/.test(src) ? src.match(/(?:embed\/|v=)([^&?/\s]+)/) : null;
|
|
533
|
+
if (ytMatch) {
|
|
534
|
+
const embedUrl = `https://www.youtube.com/watch?v=${ytMatch[1]}`;
|
|
535
|
+
const embedNode = {
|
|
536
|
+
type: 'embed',
|
|
537
|
+
text: embedUrl,
|
|
538
|
+
metadata: { embedType: 'youtube', videoId: ytMatch[1], url: embedUrl }
|
|
539
|
+
};
|
|
540
|
+
if (config.includeRawContent)
|
|
541
|
+
embedNode.rawContent = '<iframe>...</iframe>';
|
|
542
|
+
return embedNode;
|
|
543
|
+
}
|
|
544
|
+
return null;
|
|
545
|
+
}
|
|
546
|
+
// Footnotes section: its definitions were already extracted up front (see
|
|
547
|
+
// footnoteDefinitions below), so skip it here wherever it appears in the tree -
|
|
548
|
+
// it isn't necessarily a direct child of <body> (e.g. it may be nested inside
|
|
549
|
+
// a non-standalone HtmlGenerator output's wrapping <div>).
|
|
550
|
+
if (tagName === 'section' && node.attributes?.['data-footnotes'] !== undefined) {
|
|
551
|
+
return null;
|
|
552
|
+
}
|
|
553
|
+
// Math: proposed contract (no editor node built yet) - HtmlGenerator emits
|
|
554
|
+
// <span/div class="math math-inline|math-block" data-math="inline|block">
|
|
555
|
+
// with the $-delimited LaTeX as the visible (escaped) text content.
|
|
556
|
+
if ((tagName === 'div' || tagName === 'span') && node.attributes?.['data-math'] !== undefined) {
|
|
557
|
+
const mathMode = node.attributes['data-math'] === 'block' ? 'block' : 'inline';
|
|
558
|
+
const rawText = node.children.map(c => c.text || '').join('')
|
|
559
|
+
.replace(/ /g, ' ')
|
|
560
|
+
.replace(/</g, '<')
|
|
561
|
+
.replace(/>/g, '>')
|
|
562
|
+
.replace(/&/g, '&')
|
|
563
|
+
.replace(/"/g, '"')
|
|
564
|
+
.replace(/'/g, '\'');
|
|
565
|
+
const delimiter = mathMode === 'block' ? '$$' : '$';
|
|
566
|
+
const latex = rawText.startsWith(delimiter) && rawText.endsWith(delimiter)
|
|
567
|
+
? rawText.slice(delimiter.length, -delimiter.length)
|
|
568
|
+
: rawText;
|
|
569
|
+
return {
|
|
570
|
+
type: 'code',
|
|
571
|
+
text: latex,
|
|
572
|
+
metadata: { math: mathMode }
|
|
573
|
+
};
|
|
574
|
+
}
|
|
575
|
+
// Admonition: inscript-editor's Admonition node renders
|
|
576
|
+
// <div class="admonition admonition-note" data-type="note">…children…</div>.
|
|
577
|
+
if (tagName === 'div' && (node.attributes?.class || '').split(/\s+/).includes('admonition')) {
|
|
578
|
+
const admonitionTypeAttr = node.attributes?.['data-type'];
|
|
579
|
+
const admonitionType = ['note', 'tip', 'important', 'warning', 'caution'].includes(admonitionTypeAttr)
|
|
580
|
+
? admonitionTypeAttr
|
|
581
|
+
: 'note';
|
|
582
|
+
const admonitionNode = {
|
|
583
|
+
type: 'admonition',
|
|
584
|
+
metadata: { admonitionType },
|
|
585
|
+
children: parseChildren(node, newFormatting, listContext)
|
|
586
|
+
};
|
|
587
|
+
if (config.includeRawContent)
|
|
588
|
+
admonitionNode.rawContent = '<div class="admonition">...</div>';
|
|
589
|
+
return admonitionNode;
|
|
590
|
+
}
|
|
277
591
|
// Skip structural containers produced by HtmlGenerator to avoid deep AST nesting
|
|
278
592
|
if (tagName === 'div' && (node.attributes?.class === 'container' ||
|
|
279
593
|
node.attributes?.class === 'spreadsheet-container' ||
|
|
@@ -296,7 +610,7 @@ const parseHtml = async (buffer, config) => {
|
|
|
296
610
|
if (tagName === 'p' || tagName === 'div') {
|
|
297
611
|
const children = parseChildren(node, newFormatting, listContext);
|
|
298
612
|
// If it's a div and contains block elements, return children directly
|
|
299
|
-
const hasBlockElements = children.some(c => ['paragraph', 'table', 'heading', 'list', 'image', 'chart', 'code'].includes(c.type));
|
|
613
|
+
const hasBlockElements = children.some(c => ['paragraph', 'table', 'heading', 'list', 'image', 'chart', 'code', 'embed', 'admonition', 'definitionList'].includes(c.type));
|
|
300
614
|
if (tagName === 'div' && hasBlockElements) {
|
|
301
615
|
return children;
|
|
302
616
|
}
|
|
@@ -313,7 +627,8 @@ const parseHtml = async (buffer, config) => {
|
|
|
313
627
|
const pNode = {
|
|
314
628
|
type: 'paragraph',
|
|
315
629
|
metadata: { alignment: newFormatting.alignment, anchorIds: anchorIds.length > 0 ? anchorIds : undefined },
|
|
316
|
-
children: flattenedChildren
|
|
630
|
+
children: flattenedChildren,
|
|
631
|
+
htmlAttributes: collectHtmlAttributes(node, ['align'])
|
|
317
632
|
};
|
|
318
633
|
if (config.includeRawContent) {
|
|
319
634
|
// Note: Since this is a manual parser without locators, we can't easily get the original source slice.
|
|
@@ -326,17 +641,58 @@ const parseHtml = async (buffer, config) => {
|
|
|
326
641
|
const hNode = {
|
|
327
642
|
type: 'heading',
|
|
328
643
|
metadata: { level, alignment: newFormatting.alignment, anchorIds: anchorIds.length > 0 ? anchorIds : undefined },
|
|
329
|
-
children: parseChildren(node, newFormatting, listContext)
|
|
644
|
+
children: parseChildren(node, newFormatting, listContext),
|
|
645
|
+
htmlAttributes: collectHtmlAttributes(node, ['align'])
|
|
330
646
|
};
|
|
331
647
|
return hNode;
|
|
332
648
|
}
|
|
649
|
+
if (tagName === 'dl') {
|
|
650
|
+
return {
|
|
651
|
+
type: 'definitionList',
|
|
652
|
+
children: parseChildren(node, newFormatting, listContext)
|
|
653
|
+
};
|
|
654
|
+
}
|
|
655
|
+
if (tagName === 'dt') {
|
|
656
|
+
return {
|
|
657
|
+
type: 'definitionTerm',
|
|
658
|
+
children: parseChildren(node, newFormatting, listContext)
|
|
659
|
+
};
|
|
660
|
+
}
|
|
661
|
+
if (tagName === 'dd') {
|
|
662
|
+
return {
|
|
663
|
+
type: 'definitionDescription',
|
|
664
|
+
children: parseChildren(node, newFormatting, listContext)
|
|
665
|
+
};
|
|
666
|
+
}
|
|
667
|
+
if (tagName === 'abbr') {
|
|
668
|
+
const title = node.attributes?.title;
|
|
669
|
+
const children = parseChildren(node, newFormatting, listContext);
|
|
670
|
+
if (title) {
|
|
671
|
+
children.forEach(c => {
|
|
672
|
+
if (c.type === 'text') {
|
|
673
|
+
c.metadata = { ...c.metadata, abbreviationTitle: title };
|
|
674
|
+
}
|
|
675
|
+
});
|
|
676
|
+
}
|
|
677
|
+
return children;
|
|
678
|
+
}
|
|
679
|
+
if (tagName === 'cite' && node.attributes?.['data-citation-key'] !== undefined) {
|
|
680
|
+
const citationKey = node.attributes['data-citation-key'];
|
|
681
|
+
return {
|
|
682
|
+
type: 'text',
|
|
683
|
+
text: citationKey,
|
|
684
|
+
formatting: Object.keys(newFormatting).length > 0 ? { ...newFormatting } : undefined,
|
|
685
|
+
metadata: { citationKey }
|
|
686
|
+
};
|
|
687
|
+
}
|
|
333
688
|
if (tagName === 'ul' || tagName === 'ol') {
|
|
334
689
|
const isNewTopLevel = !listContext;
|
|
335
690
|
const newListContext = {
|
|
336
691
|
listId: isNewTopLevel ? `html-list-${htmlListIdCounter++}` : listContext.listId,
|
|
337
692
|
type: tagName === 'ol' ? 'ordered' : 'unordered',
|
|
338
693
|
level: isNewTopLevel ? 0 : listContext.level + 1,
|
|
339
|
-
counters: isNewTopLevel ? {} : { ...listContext.counters } // Clone to avoid side effects on parent levels
|
|
694
|
+
counters: isNewTopLevel ? {} : { ...listContext.counters }, // Clone to avoid side effects on parent levels
|
|
695
|
+
isTask: node.attributes?.['data-type'] === 'taskList'
|
|
340
696
|
};
|
|
341
697
|
// Initialize counter for this level
|
|
342
698
|
if (tagName === 'ol' && node.attributes?.start) {
|
|
@@ -362,6 +718,13 @@ const parseHtml = async (buffer, config) => {
|
|
|
362
718
|
const children = parseChildren(node, newFormatting, listContext);
|
|
363
719
|
const nestedLists = children.filter(c => c.type === 'list');
|
|
364
720
|
const selfChildren = children.filter(c => c.type !== 'list');
|
|
721
|
+
let isTask;
|
|
722
|
+
let checked;
|
|
723
|
+
if (listContext?.isTask) {
|
|
724
|
+
isTask = true;
|
|
725
|
+
const dataChecked = node.attributes?.['data-checked'];
|
|
726
|
+
checked = dataChecked !== undefined ? dataChecked === 'true' : (findNestedCheckboxChecked(node) ?? false);
|
|
727
|
+
}
|
|
365
728
|
const selfNode = {
|
|
366
729
|
type: 'list',
|
|
367
730
|
text: selfChildren.map(c => c.text || '').join(''),
|
|
@@ -371,17 +734,23 @@ const parseHtml = async (buffer, config) => {
|
|
|
371
734
|
alignment: newFormatting.alignment || 'left',
|
|
372
735
|
listId: listContext?.listId || 'html-list-none',
|
|
373
736
|
itemIndex: (listContext?.counters[listContext.level] ?? 1) - 1,
|
|
374
|
-
anchorIds: anchorIds.length > 0 ? anchorIds : undefined
|
|
737
|
+
anchorIds: anchorIds.length > 0 ? anchorIds : undefined,
|
|
738
|
+
isTask,
|
|
739
|
+
checked
|
|
375
740
|
},
|
|
376
741
|
children: selfChildren
|
|
377
742
|
};
|
|
378
743
|
return [selfNode, ...nestedLists];
|
|
379
744
|
}
|
|
380
745
|
if (tagName === 'table') {
|
|
746
|
+
// CustomTable (inscript-editor) renders data-align on the <table> itself.
|
|
747
|
+
const tableAlignAttr = node.attributes?.['data-align'];
|
|
748
|
+
const tableAlign = ['left', 'center', 'right'].includes(tableAlignAttr) ? tableAlignAttr : undefined;
|
|
381
749
|
const tableNode = {
|
|
382
750
|
type: 'table',
|
|
383
|
-
metadata: { anchorIds: anchorIds.length > 0 ? anchorIds : undefined },
|
|
384
|
-
children: parseChildren(node, newFormatting, listContext)
|
|
751
|
+
metadata: { anchorIds: anchorIds.length > 0 ? anchorIds : undefined, align: tableAlign },
|
|
752
|
+
children: parseChildren(node, newFormatting, listContext),
|
|
753
|
+
htmlAttributes: collectHtmlAttributes(node, ['data-align', 'align'])
|
|
385
754
|
};
|
|
386
755
|
if (config.includeRawContent) {
|
|
387
756
|
tableNode.rawContent = '<table>...</table>';
|
|
@@ -391,7 +760,8 @@ const parseHtml = async (buffer, config) => {
|
|
|
391
760
|
if (tagName === 'tr') {
|
|
392
761
|
const rowNode = {
|
|
393
762
|
type: 'row',
|
|
394
|
-
children: parseChildren(node, newFormatting, listContext)
|
|
763
|
+
children: parseChildren(node, newFormatting, listContext),
|
|
764
|
+
htmlAttributes: collectHtmlAttributes(node, [])
|
|
395
765
|
};
|
|
396
766
|
if (config.includeRawContent) {
|
|
397
767
|
rowNode.rawContent = '<tr>...</tr>';
|
|
@@ -399,9 +769,20 @@ const parseHtml = async (buffer, config) => {
|
|
|
399
769
|
return rowNode;
|
|
400
770
|
}
|
|
401
771
|
if (tagName === 'td' || tagName === 'th') {
|
|
772
|
+
// Merged cells: mirrors the colspan/rowspan reading already done in
|
|
773
|
+
// MarkdownParser's inline HTML-table handler.
|
|
774
|
+
const colSpanAttr = node.attributes?.colspan;
|
|
775
|
+
const rowSpanAttr = node.attributes?.rowspan;
|
|
776
|
+
const colSpan = colSpanAttr ? parseInt(colSpanAttr, 10) : undefined;
|
|
777
|
+
const rowSpan = rowSpanAttr ? parseInt(rowSpanAttr, 10) : undefined;
|
|
402
778
|
const cellNode = {
|
|
403
779
|
type: 'cell',
|
|
404
|
-
|
|
780
|
+
metadata: {
|
|
781
|
+
colSpan: colSpan && !isNaN(colSpan) ? colSpan : undefined,
|
|
782
|
+
rowSpan: rowSpan && !isNaN(rowSpan) ? rowSpan : undefined
|
|
783
|
+
},
|
|
784
|
+
children: parseChildren(node, newFormatting, listContext),
|
|
785
|
+
htmlAttributes: collectHtmlAttributes(node, ['colspan', 'rowspan', 'align'])
|
|
405
786
|
};
|
|
406
787
|
if (config.includeRawContent) {
|
|
407
788
|
cellNode.rawContent = '<td>...</td>';
|
|
@@ -411,6 +792,30 @@ const parseHtml = async (buffer, config) => {
|
|
|
411
792
|
if (tagName === 'img') {
|
|
412
793
|
const src = node.attributes?.src;
|
|
413
794
|
const alt = node.attributes?.alt;
|
|
795
|
+
// CustomImage (inscript-editor) renders data-width/data-align, falling back to
|
|
796
|
+
// parsing the inline style for consumers that only emit the CSS.
|
|
797
|
+
const imgDecls = parseStyleDeclarations(node.attributes?.style || '');
|
|
798
|
+
// Exact lookup, so `max-width: 100%` - the standard responsive-image style, and by
|
|
799
|
+
// far the most common inline style on an <img> - is no longer read as a declared
|
|
800
|
+
// width. It constrains the rendered size; it is not an author-specified width.
|
|
801
|
+
const width = node.attributes?.['data-width'] || getDeclaration(imgDecls, 'width');
|
|
802
|
+
// Alignment is inferred from which auto margin is present. Comparing the parsed
|
|
803
|
+
// value rather than substring-matching "margin-left: 0" stops `margin-left: 0.5rem`
|
|
804
|
+
// from being read as left-aligned, and lets the `margin: 0 auto` centering
|
|
805
|
+
// shorthand be recognised at all.
|
|
806
|
+
const marginLeft = getDeclaration(imgDecls, 'margin-left');
|
|
807
|
+
const marginRight = getDeclaration(imgDecls, 'margin-right');
|
|
808
|
+
const marginShorthand = getDeclaration(imgDecls, 'margin');
|
|
809
|
+
const isZero = (v) => v !== undefined && /^0(?:[a-z%]*)$/.test(v);
|
|
810
|
+
const shorthandParts = marginShorthand ? marginShorthand.split(/\s+/) : [];
|
|
811
|
+
const shorthandCentres = shorthandParts.length > 1
|
|
812
|
+
&& shorthandParts[shorthandParts.length - 1] === 'auto'
|
|
813
|
+
&& shorthandParts[1] === 'auto';
|
|
814
|
+
const alignAttr = node.attributes?.['data-align']
|
|
815
|
+
?? (shorthandCentres ? 'center'
|
|
816
|
+
: (isZero(marginLeft) && !isZero(marginRight) ? 'left'
|
|
817
|
+
: (isZero(marginRight) && !isZero(marginLeft) ? 'right' : undefined)));
|
|
818
|
+
const align = ['left', 'center', 'right'].includes(alignAttr) ? alignAttr : undefined;
|
|
414
819
|
let imageNode;
|
|
415
820
|
if (src?.startsWith('data:')) {
|
|
416
821
|
const match = src.match(/^data:([^;]+);base64,(.*)$/);
|
|
@@ -429,7 +834,9 @@ const parseHtml = async (buffer, config) => {
|
|
|
429
834
|
type: 'image',
|
|
430
835
|
metadata: {
|
|
431
836
|
attachmentName: name,
|
|
432
|
-
altText: alt
|
|
837
|
+
altText: alt,
|
|
838
|
+
width,
|
|
839
|
+
align
|
|
433
840
|
}
|
|
434
841
|
};
|
|
435
842
|
}
|
|
@@ -438,7 +845,9 @@ const parseHtml = async (buffer, config) => {
|
|
|
438
845
|
type: 'image',
|
|
439
846
|
metadata: {
|
|
440
847
|
url: src,
|
|
441
|
-
altText: alt
|
|
848
|
+
altText: alt,
|
|
849
|
+
width,
|
|
850
|
+
align
|
|
442
851
|
}
|
|
443
852
|
};
|
|
444
853
|
}
|
|
@@ -449,7 +858,9 @@ const parseHtml = async (buffer, config) => {
|
|
|
449
858
|
metadata: {
|
|
450
859
|
url: src,
|
|
451
860
|
altText: alt,
|
|
452
|
-
anchorIds: anchorIds.length > 0 ? anchorIds : undefined
|
|
861
|
+
anchorIds: anchorIds.length > 0 ? anchorIds : undefined,
|
|
862
|
+
width,
|
|
863
|
+
align
|
|
453
864
|
}
|
|
454
865
|
};
|
|
455
866
|
}
|
|
@@ -460,8 +871,16 @@ const parseHtml = async (buffer, config) => {
|
|
|
460
871
|
}
|
|
461
872
|
if (tagName === 'a') {
|
|
462
873
|
const href = node.attributes?.href;
|
|
874
|
+
const wikilinkPage = node.attributes?.['data-wikilink-page'];
|
|
463
875
|
const children = parseChildren(node, newFormatting, listContext);
|
|
464
|
-
if (
|
|
876
|
+
if (wikilinkPage !== undefined) {
|
|
877
|
+
children.forEach(c => {
|
|
878
|
+
if (c.type === 'text') {
|
|
879
|
+
c.metadata = { ...c.metadata, link: wikilinkPage, linkType: 'internal', wikilink: true };
|
|
880
|
+
}
|
|
881
|
+
});
|
|
882
|
+
}
|
|
883
|
+
else if (href) {
|
|
465
884
|
const linkType = href.startsWith('#') ? 'internal' : 'external';
|
|
466
885
|
children.forEach(c => {
|
|
467
886
|
if (c.type === 'text') {
|
|
@@ -509,6 +928,42 @@ const parseHtml = async (buffer, config) => {
|
|
|
509
928
|
}
|
|
510
929
|
return null;
|
|
511
930
|
};
|
|
931
|
+
// Extract <section data-footnotes> up front so its definitions are available to
|
|
932
|
+
// <sup data-footnote-ref> references encountered anywhere earlier in the body.
|
|
933
|
+
const findFootnotesSection = (n) => {
|
|
934
|
+
if (n.tagName === 'section' && n.attributes?.['data-footnotes'] !== undefined)
|
|
935
|
+
return n;
|
|
936
|
+
for (const child of n.children) {
|
|
937
|
+
const found = findFootnotesSection(child);
|
|
938
|
+
if (found)
|
|
939
|
+
return found;
|
|
940
|
+
}
|
|
941
|
+
return undefined;
|
|
942
|
+
};
|
|
943
|
+
const footnotesSectionNode = findFootnotesSection(body);
|
|
944
|
+
if (footnotesSectionNode) {
|
|
945
|
+
for (const item of footnotesSectionNode.children) {
|
|
946
|
+
if (item.type !== 'element')
|
|
947
|
+
continue;
|
|
948
|
+
const key = item.attributes?.['data-footnote-id'];
|
|
949
|
+
if (!key)
|
|
950
|
+
continue;
|
|
951
|
+
// Strip the generated back-reference link ("↩") - it's round-trip plumbing,
|
|
952
|
+
// not part of the footnote's actual content.
|
|
953
|
+
const filteredChildren = item.children.filter(c => !(c.tagName === 'a' && (c.attributes?.href || '').startsWith('#footnote-ref-')));
|
|
954
|
+
const contentNodes = [];
|
|
955
|
+
for (const child of filteredChildren) {
|
|
956
|
+
const parsed = parseNode(child);
|
|
957
|
+
if (parsed) {
|
|
958
|
+
if (Array.isArray(parsed))
|
|
959
|
+
contentNodes.push(...parsed);
|
|
960
|
+
else
|
|
961
|
+
contentNodes.push(parsed);
|
|
962
|
+
}
|
|
963
|
+
}
|
|
964
|
+
footnoteDefinitions.set(key, contentNodes);
|
|
965
|
+
}
|
|
966
|
+
}
|
|
512
967
|
for (const child of body.children) {
|
|
513
968
|
const parsed = parseNode(child);
|
|
514
969
|
if (parsed) {
|
|
@@ -539,8 +994,12 @@ const parseHtml = async (buffer, config) => {
|
|
|
539
994
|
return node.text || '';
|
|
540
995
|
if (node.type === 'break')
|
|
541
996
|
return '\n';
|
|
997
|
+
// Childless nodes still carry meaningful text - fall back to it instead of
|
|
998
|
+
// silently vanishing from plain-text/RAG-chunk output.
|
|
999
|
+
if (node.type === 'embed')
|
|
1000
|
+
return node.metadata?.url || '';
|
|
542
1001
|
if (node.children) {
|
|
543
|
-
const isBlock = ['table', 'row', 'list', 'sheet', 'slide'].includes(node.type);
|
|
1002
|
+
const isBlock = ['table', 'row', 'list', 'sheet', 'slide', 'admonition', 'definitionList'].includes(node.type);
|
|
544
1003
|
return node.children.map(getText).join(isBlock ? config.newlineDelimiter : '');
|
|
545
1004
|
}
|
|
546
1005
|
return '';
|