officeparser 7.2.3 → 7.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/README.md +277 -17
  2. package/dist/OfficeConverter.d.ts +1 -1
  3. package/dist/OfficeConverter.js +3 -0
  4. package/dist/OfficeGenerator.js +4 -0
  5. package/dist/OfficeParser.d.ts +2 -0
  6. package/dist/OfficeParser.js +6 -0
  7. package/dist/cli.d.ts +1 -1
  8. package/dist/cli.js +5 -2
  9. package/dist/defaults.js +12 -0
  10. package/dist/generators/BaseGenerator.d.ts +34 -1
  11. package/dist/generators/BaseGenerator.js +98 -0
  12. package/dist/generators/CsvGenerator.d.ts +9 -1
  13. package/dist/generators/CsvGenerator.js +28 -16
  14. package/dist/generators/EpubGenerator.d.ts +43 -0
  15. package/dist/generators/EpubGenerator.js +312 -0
  16. package/dist/generators/HtmlGenerator.d.ts +12 -0
  17. package/dist/generators/HtmlGenerator.js +378 -61
  18. package/dist/generators/MarkdownGenerator.d.ts +28 -5
  19. package/dist/generators/MarkdownGenerator.js +432 -51
  20. package/dist/generators/PdfGenerator.js +32 -0
  21. package/dist/generators/RtfGenerator.js +47 -22
  22. package/dist/generators/TextGenerator.js +98 -11
  23. package/dist/index.d.ts +1 -0
  24. package/dist/index.js +1 -0
  25. package/dist/officeparser.browser.d.ts +427 -20
  26. package/dist/officeparser.browser.iife.js +338 -206
  27. package/dist/officeparser.browser.mjs +346 -214
  28. package/dist/officeparser.browser.slim.d.ts +427 -20
  29. package/dist/officeparser.browser.slim.iife.js +346 -214
  30. package/dist/officeparser.browser.slim.mjs +346 -214
  31. package/dist/parsers/EpubParser.d.ts +8 -0
  32. package/dist/parsers/EpubParser.js +217 -0
  33. package/dist/parsers/ExcelParser.js +2 -0
  34. package/dist/parsers/HtmlParser.js +507 -48
  35. package/dist/parsers/MarkdownParser.js +704 -92
  36. package/dist/parsers/OpenOfficeParser.js +128 -20
  37. package/dist/parsers/PdfParser.js +4 -1
  38. package/dist/parsers/PowerPointParser.js +1 -0
  39. package/dist/parsers/WordParser.js +1 -0
  40. package/dist/sbom.cdx.json +1695 -0
  41. package/dist/types.d.ts +427 -20
  42. package/dist/types.js +8 -0
  43. package/dist/utils/configUtils.js +53 -4
  44. package/dist/utils/errorUtils.js +7 -3
  45. package/dist/utils/sanitize.d.ts +139 -0
  46. package/dist/utils/sanitize.js +318 -0
  47. package/dist/utils/xmlUtils.js +2 -2
  48. package/dist/utils/zipUtils.js +76 -26
  49. package/package.json +16 -12
@@ -1,11 +1,25 @@
1
1
  "use strict";
2
2
  Object.defineProperty(exports, "__esModule", { value: true });
3
3
  exports.parseHtml = void 0;
4
+ const types_js_1 = require("../types.js");
4
5
  const astUtils_js_1 = require("../utils/astUtils.js");
5
6
  const errorUtils_js_1 = require("../utils/errorUtils.js");
7
+ const sanitize_js_1 = require("../utils/sanitize.js");
8
+ /**
9
+ * Maximum element nesting depth accepted from an HTML/XHTML source before the parser gives up
10
+ * with a typed error rather than letting the recursion overflow the call stack. See the guard in
11
+ * `parseNode` for why this value and not a larger one.
12
+ */
13
+ const MAX_HTML_NESTING_DEPTH = 256;
6
14
  const parseAttributes = (attrString) => {
7
15
  const attrs = {};
8
- const regex = /([a-zA-Z0-9\-:]+)(?:\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s>]+)))?/g;
16
+ // Attribute names follow the HTML5 rule - any character except whitespace and
17
+ // " ' > / = - rather than a hand-picked allowlist. The previous class
18
+ // ([a-zA-Z0-9\-:]) silently split a legal name on any character outside it, so
19
+ // `data_foo="x"` produced TWO attributes: `data` (empty) and an invented
20
+ // `foo="x"` that was never in the source. Harmless while nothing read unknown
21
+ // attributes; not harmless once they can be replayed into generated output.
22
+ const regex = /([^\s"'>/=]+)(?:\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s>]+)))?/g;
9
23
  let match;
10
24
  while ((match = regex.exec(attrString)) !== null) {
11
25
  const name = match[1].toLowerCase();
@@ -14,6 +28,101 @@ const parseAttributes = (attrString) => {
14
28
  }
15
29
  return attrs;
16
30
  };
31
+ /**
32
+ * Splits an inline `style` attribute into a property -> value map.
33
+ *
34
+ * Replaces substring matching (`styleAttr.includes('font-weight: bold')`) and unanchored regexes
35
+ * (`/color:\s*([^;]+)/`), which were wrong in both directions:
36
+ * - false positives: `color:` matched inside `background-color:`, so
37
+ * `"background-color: red; color: blue"` yielded color=red - the *wrong* value, not merely a
38
+ * spurious one - and `width:` matched inside `max-width:`, so the ubiquitous responsive-image
39
+ * style `max-width: 100%` was read as an explicit width.
40
+ * - false negatives: `font-weight:bold` without a space, `font-weight: 700`, and `bolder` were
41
+ * all missed, as was `line-through` inside `text-decoration: underline line-through`.
42
+ *
43
+ * Splitting is quote- and paren-aware so a semicolon inside `url(data:image/png;base64,...)` or a
44
+ * quoted font stack doesn't shatter the declaration. `!important` is stripped from values, since
45
+ * substring matching used to tolerate it and exact comparison otherwise would not - dropping it
46
+ * would be a silent regression rather than the intended fix.
47
+ */
48
+ const parseStyleDeclarations = (styleAttr) => {
49
+ const decls = new Map();
50
+ if (!styleAttr)
51
+ return decls;
52
+ let depth = 0;
53
+ let quote = null;
54
+ let current = '';
55
+ const chunks = [];
56
+ for (const ch of styleAttr) {
57
+ if (quote) {
58
+ if (ch === quote)
59
+ quote = null;
60
+ }
61
+ else if (ch === '"' || ch === '\'') {
62
+ quote = ch;
63
+ }
64
+ else if (ch === '(') {
65
+ depth++;
66
+ }
67
+ else if (ch === ')') {
68
+ if (depth > 0)
69
+ depth--;
70
+ }
71
+ else if (ch === ';' && depth === 0) {
72
+ chunks.push(current);
73
+ current = '';
74
+ continue;
75
+ }
76
+ current += ch;
77
+ }
78
+ chunks.push(current);
79
+ for (const chunk of chunks) {
80
+ const idx = chunk.indexOf(':');
81
+ if (idx === -1)
82
+ continue;
83
+ const prop = chunk.slice(0, idx).trim().toLowerCase();
84
+ if (!prop)
85
+ continue;
86
+ const value = chunk.slice(idx + 1).trim().replace(/\s*!\s*important\s*$/i, '').trim();
87
+ if (value)
88
+ decls.set(prop, value);
89
+ }
90
+ return decls;
91
+ };
92
+ /**
93
+ * Reads a declaration, also accepting the `-webkit-`/`-moz-`/`-ms-`/`-o-` prefixed spelling so a
94
+ * vendor-prefixed property keeps matching (substring matching used to catch those by accident).
95
+ */
96
+ const getDeclaration = (decls, prop) => decls.get(prop)
97
+ ?? decls.get(`-webkit-${prop}`)
98
+ ?? decls.get(`-moz-${prop}`)
99
+ ?? decls.get(`-ms-${prop}`)
100
+ ?? decls.get(`-o-${prop}`);
101
+ /**
102
+ * Returns the first family from a `font-family` stack, respecting quotes so a quoted family name
103
+ * containing a comma (`'Fira, A', serif`) isn't split through the middle of its own name.
104
+ */
105
+ const firstFontFamily = (fontFamily) => {
106
+ let quote = null;
107
+ let first = '';
108
+ for (const ch of fontFamily) {
109
+ if (quote) {
110
+ if (ch === quote) {
111
+ quote = null;
112
+ continue;
113
+ }
114
+ }
115
+ else if (ch === '"' || ch === '\'') {
116
+ quote = ch;
117
+ continue;
118
+ }
119
+ else if (ch === ',') {
120
+ break;
121
+ }
122
+ first += ch;
123
+ }
124
+ return first.trim();
125
+ };
17
126
  const parseHtmlTree = (html) => {
18
127
  const root = { type: 'element', tagName: 'root', children: [], attributes: {} };
19
128
  let current = root;
@@ -36,14 +145,16 @@ const parseHtmlTree = (html) => {
36
145
  cursor = commentEnd !== -1 ? commentEnd + 3 : html.length;
37
146
  continue;
38
147
  }
39
- const tagEndMatch = html.substring(tagStart).match(/>/);
40
- if (!tagEndMatch) {
148
+ // indexOf (not substring().match) so scanning for the tag end is O(1) in
149
+ // allocation — a document with many "<" chars would otherwise be O(n^2).
150
+ const tagEndIdx = html.indexOf('>', tagStart);
151
+ if (tagEndIdx === -1) {
41
152
  const text = html.substring(tagStart);
42
153
  current.children.push({ type: 'text', text, children: [], parent: current });
43
154
  break;
44
155
  }
45
- const tagContent = html.substring(tagStart + 1, tagStart + tagEndMatch.index);
46
- cursor = tagStart + tagEndMatch.index + 1;
156
+ const tagContent = html.substring(tagStart + 1, tagEndIdx);
157
+ cursor = tagEndIdx + 1;
47
158
  const isClosing = tagContent.startsWith('/');
48
159
  const isSelfClosing = tagContent.endsWith('/');
49
160
  const tagCore = tagContent.replace(/^\/|\/$/g, '').trim();
@@ -77,16 +188,20 @@ const parseHtmlTree = (html) => {
77
188
  if (!isSelfClosing && !voidElements.has(tagName)) {
78
189
  current = node;
79
190
  if (tagName === 'script' || tagName === 'style') {
80
- const closeTag = `</${tagName}>`;
81
- const closeIdx = html.toLowerCase().indexOf(closeTag, cursor);
82
- if (closeIdx !== -1) {
191
+ // Case-insensitive search from `cursor` via a sticky-ish regex, instead of
192
+ // lower-casing the whole document on every <script>/<style> (was O(n^2)).
193
+ // tagName is validated to /^[a-z0-9-]+$/ above, so it's safe to interpolate.
194
+ const closeRe = new RegExp(`</${tagName}>`, 'gi');
195
+ closeRe.lastIndex = cursor;
196
+ const closeMatch = closeRe.exec(html);
197
+ if (closeMatch) {
83
198
  node.children.push({
84
199
  type: 'text',
85
- text: html.substring(cursor, closeIdx),
200
+ text: html.substring(cursor, closeMatch.index),
86
201
  children: [],
87
202
  parent: node
88
203
  });
89
- cursor = closeIdx + closeTag.length;
204
+ cursor = closeMatch.index + closeMatch[0].length;
90
205
  current = node.parent;
91
206
  }
92
207
  }
@@ -183,7 +298,82 @@ const parseHtml = async (buffer, config) => {
183
298
  }
184
299
  const content = [];
185
300
  let htmlListIdCounter = 1;
186
- const parseNode = (node, currentFormatting = {}, listContext) => {
301
+ // Finds the checked state from a nested <input type="checkbox"> (GFM task-list items
302
+ // nest it inside a <label>, so it isn't a direct child of the <li>).
303
+ const findNestedCheckboxChecked = (n) => {
304
+ if (n.tagName === 'input' && (n.attributes?.type || '').toLowerCase() === 'checkbox') {
305
+ return 'checked' in (n.attributes || {});
306
+ }
307
+ for (const child of n.children) {
308
+ const found = findNestedCheckboxChecked(child);
309
+ if (found !== undefined)
310
+ return found;
311
+ }
312
+ return undefined;
313
+ };
314
+ // Populated from a <section data-footnotes> block (found and parsed before the main
315
+ // body loop, since references can appear anywhere earlier in the document) and
316
+ // consulted by parseChildren's <sup data-footnote-ref> handling below.
317
+ const footnoteDefinitions = new Map();
318
+ // --- Generic attribute pass-through (htmlParserConfig.preserveAttributes) ---------------
319
+ // Captures attributes no typed metadata field consumed, so they can be replayed on
320
+ // generation. Everything here is a *defence-in-depth* filter: HtmlGenerator sanitizes again
321
+ // on the way out, because an AST can be built programmatically rather than parsed.
322
+ const preserveAttributes = config.htmlParserConfig?.preserveAttributes === true;
323
+ // `style` is already consumed wholesale into TextFormatting/metadata above, and `id` is
324
+ // consumed into anchorIds and re-emitted by the generator - carrying either would duplicate
325
+ // an attribute the generator composes itself. `class` is deliberately NOT excluded: the
326
+ // generator's class attribute is built purely from style-mapping and never from a parsed
327
+ // `class`, so without this a plain `<p class="lead">` loses "lead" entirely. The generator
328
+ // merges it into that attribute rather than emitting a second one.
329
+ const GENERATOR_OWNED_ATTRS = new Set(['id', 'style']);
330
+ /**
331
+ * Captures the attributes of `node` that `consumed` didn't claim.
332
+ * Returns undefined when nothing survives, so the field stays absent rather than `{}`.
333
+ */
334
+ const collectHtmlAttributes = (node, consumed) => {
335
+ if (!preserveAttributes || !node.attributes)
336
+ return undefined;
337
+ const consumedSet = new Set(consumed.map(c => c.toLowerCase()));
338
+ const bag = {};
339
+ for (const [rawKey, value] of Object.entries(node.attributes)) {
340
+ const key = rawKey.toLowerCase();
341
+ if (consumedSet.has(key) || GENERATOR_OWNED_ATTRS.has(key))
342
+ continue;
343
+ // Event handlers are never carried, at any layer, with no opt-in.
344
+ if (/^on/i.test(key))
345
+ continue;
346
+ // srcdoc holds a whole HTML document; it cannot be safely escaped into an attribute.
347
+ if (key === 'srcdoc')
348
+ continue;
349
+ // Reject anything that isn't a plain attribute name outright - a key containing a
350
+ // quote or '=' is the shape an attribute-injection payload takes.
351
+ if (!(0, sanitize_js_1.isSafeHtmlAttributeName)(key))
352
+ continue;
353
+ bag[key] = value;
354
+ }
355
+ return Object.keys(bag).length > 0 ? bag : undefined;
356
+ };
357
+ const parseNode = (node, currentFormatting = {}, listContext, depth = 0) => {
358
+ // Guard against a maliciously deep element tree (e.g. tens of thousands of nested
359
+ // <div>) recursing until the call stack overflows.
360
+ //
361
+ // The previous limit of 1000 could never fire: measured overflow is around 800 and
362
+ // varies run to run (796/862/796 on three identical runs), so the RangeError always
363
+ // arrived first and the typed error this guard exists to produce never did. Failure was
364
+ // still graceful - it surfaces as a wrapped Error, not a crash - which is why this was a
365
+ // dead guard rather than a denial of service.
366
+ //
367
+ // 256 is chosen to hold across engines rather than tuned to one. It is far below the
368
+ // lowest overflow observed here and leaves room for a smaller frame budget on older V8
369
+ // (the supported floor is Node 18), while sitting orders of magnitude above real
370
+ // content: the bundled HTML and EPUB fixtures reach an AST depth of 8.
371
+ // Per node, alongside the depth guard: the two together are what make a hostile
372
+ // document both bounded and cancellable rather than only bounded.
373
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
374
+ if (depth > MAX_HTML_NESTING_DEPTH) {
375
+ throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.MAX_NESTING_DEPTH_EXCEEDED);
376
+ }
187
377
  if (node.type === 'text') {
188
378
  let decodedText = (node.text || '')
189
379
  .replace(/&nbsp;/g, ' ')
@@ -229,29 +419,43 @@ const parseHtml = async (buffer, config) => {
229
419
  const styleAttr = node.attributes?.style || '';
230
420
  const alignAttr = node.attributes?.align || '';
231
421
  if (styleAttr || alignAttr) {
232
- if (styleAttr.includes('font-weight: bold'))
233
- newFormatting.bold = true;
234
- if (styleAttr.includes('font-style: italic'))
422
+ const decls = parseStyleDeclarations(styleAttr);
423
+ // `bold`, `bolder`, and any weight >= 600 are all bold; the old substring check
424
+ // only ever saw the literal "font-weight: bold".
425
+ const weight = getDeclaration(decls, 'font-weight');
426
+ if (weight) {
427
+ const numericWeight = parseInt(weight, 10);
428
+ if (weight === 'bold' || weight === 'bolder' || (!isNaN(numericWeight) && numericWeight >= 600)) {
429
+ newFormatting.bold = true;
430
+ }
431
+ }
432
+ if (getDeclaration(decls, 'font-style') === 'italic')
235
433
  newFormatting.italic = true;
236
- if (styleAttr.includes('text-decoration: underline'))
237
- newFormatting.underline = true;
238
- if (styleAttr.includes('text-decoration: line-through'))
239
- newFormatting.strikethrough = true;
240
- const colorMatch = styleAttr.match(/color:\s*([^;]+)/);
241
- if (colorMatch)
242
- newFormatting.color = colorMatch[1].trim();
243
- const bgMatch = styleAttr.match(/background-color:\s*([^;]+)/);
244
- if (bgMatch)
245
- newFormatting.backgroundColor = bgMatch[1].trim();
246
- const sizeMatch = styleAttr.match(/font-size:\s*([^;]+)/);
247
- if (sizeMatch)
248
- newFormatting.size = sizeMatch[1].trim();
249
- const fontMatch = styleAttr.match(/font-family:\s*([^;]+)/);
250
- if (fontMatch)
251
- newFormatting.font = fontMatch[1].trim().split(',')[0].replace(/['"]/g, '');
252
- const alignmentMatch = styleAttr.match(/text-align:\s*(left|center|right|justify)/);
253
- if (alignmentMatch) {
254
- newFormatting.alignment = alignmentMatch[1].toLowerCase();
434
+ // text-decoration is a shorthand that can carry several keywords at once, so
435
+ // "underline line-through" has to set both flags rather than only the first.
436
+ const decoration = getDeclaration(decls, 'text-decoration') ?? getDeclaration(decls, 'text-decoration-line');
437
+ if (decoration) {
438
+ const parts = decoration.split(/\s+/);
439
+ if (parts.includes('underline'))
440
+ newFormatting.underline = true;
441
+ if (parts.includes('line-through'))
442
+ newFormatting.strikethrough = true;
443
+ }
444
+ const color = decls.get('color');
445
+ if (color)
446
+ newFormatting.color = color;
447
+ const background = getDeclaration(decls, 'background-color');
448
+ if (background)
449
+ newFormatting.backgroundColor = background;
450
+ const size = getDeclaration(decls, 'font-size');
451
+ if (size)
452
+ newFormatting.size = size;
453
+ const fontFamily = getDeclaration(decls, 'font-family');
454
+ if (fontFamily)
455
+ newFormatting.font = firstFontFamily(fontFamily);
456
+ const textAlign = getDeclaration(decls, 'text-align')?.toLowerCase();
457
+ if (textAlign && ['left', 'center', 'right', 'justify'].includes(textAlign)) {
458
+ newFormatting.alignment = textAlign;
255
459
  }
256
460
  else if (alignAttr) {
257
461
  const align = alignAttr.toLowerCase();
@@ -264,7 +468,29 @@ const parseHtml = async (buffer, config) => {
264
468
  const parseChildren = (n, fmt, lCtx) => {
265
469
  const kids = [];
266
470
  for (const child of n.children) {
267
- const parsed = parseNode(child, fmt, lCtx);
471
+ // Footnote/endnote reference: attach as .notes on the preceding node
472
+ // instead of inserting a visible node, matching WordParser's convention.
473
+ if (child.type === 'element' && child.tagName === 'sup' && child.attributes?.['data-footnote-ref'] !== undefined) {
474
+ const key = child.attributes['data-footnote-ref'];
475
+ const definition = footnoteDefinitions.get(key);
476
+ const noteNode = {
477
+ type: 'note',
478
+ text: (definition || []).map(d => d.text || '').join(''),
479
+ children: definition || [],
480
+ metadata: { noteType: 'footnote', noteId: key }
481
+ };
482
+ if (kids.length > 0) {
483
+ const target = kids[kids.length - 1];
484
+ if (!target.notes)
485
+ target.notes = [];
486
+ target.notes.push(noteNode);
487
+ }
488
+ else {
489
+ kids.push({ type: 'text', text: '', notes: [noteNode] });
490
+ }
491
+ continue;
492
+ }
493
+ const parsed = parseNode(child, fmt, lCtx, depth + 1);
268
494
  if (parsed) {
269
495
  if (Array.isArray(parsed))
270
496
  kids.push(...parsed);
@@ -274,6 +500,94 @@ const parseHtml = async (buffer, config) => {
274
500
  }
275
501
  return kids;
276
502
  };
503
+ // YouTube embeds: inscript-editor's Youtube node renders
504
+ // <div data-youtube-video="ID" data-width="…" data-align="…">…<iframe…></div>.
505
+ // Recognise both the wrapper div and a bare iframe so externally-authored HTML
506
+ // (and a saved-then-reopened .md that fell back to raw HTML) both round-trip.
507
+ if (tagName === 'div' && node.attributes?.['data-youtube-video'] !== undefined) {
508
+ const videoId = node.attributes['data-youtube-video'] || '';
509
+ const width = node.attributes?.['data-width'];
510
+ const embedAlignAttr = node.attributes?.['data-align'];
511
+ const embedAlign = ['left', 'center', 'right'].includes(embedAlignAttr) ? embedAlignAttr : undefined;
512
+ const embedUrl = videoId ? `https://www.youtube.com/watch?v=${videoId}` : undefined;
513
+ const embedNode = {
514
+ type: 'embed',
515
+ // Childless nodes need .text so generic AST consumers (toText, chunking)
516
+ // don't silently drop them.
517
+ text: embedUrl,
518
+ metadata: {
519
+ embedType: 'youtube',
520
+ videoId,
521
+ url: embedUrl,
522
+ width,
523
+ align: embedAlign
524
+ }
525
+ };
526
+ if (config.includeRawContent)
527
+ embedNode.rawContent = '<div data-youtube-video>...</div>';
528
+ return embedNode;
529
+ }
530
+ if (tagName === 'iframe') {
531
+ const src = node.attributes?.src || '';
532
+ const ytMatch = /youtube(?:-nocookie)?\.com/.test(src) ? src.match(/(?:embed\/|v=)([^&?/\s]+)/) : null;
533
+ if (ytMatch) {
534
+ const embedUrl = `https://www.youtube.com/watch?v=${ytMatch[1]}`;
535
+ const embedNode = {
536
+ type: 'embed',
537
+ text: embedUrl,
538
+ metadata: { embedType: 'youtube', videoId: ytMatch[1], url: embedUrl }
539
+ };
540
+ if (config.includeRawContent)
541
+ embedNode.rawContent = '<iframe>...</iframe>';
542
+ return embedNode;
543
+ }
544
+ return null;
545
+ }
546
+ // Footnotes section: its definitions were already extracted up front (see
547
+ // footnoteDefinitions below), so skip it here wherever it appears in the tree -
548
+ // it isn't necessarily a direct child of <body> (e.g. it may be nested inside
549
+ // a non-standalone HtmlGenerator output's wrapping <div>).
550
+ if (tagName === 'section' && node.attributes?.['data-footnotes'] !== undefined) {
551
+ return null;
552
+ }
553
+ // Math: proposed contract (no editor node built yet) - HtmlGenerator emits
554
+ // <span/div class="math math-inline|math-block" data-math="inline|block">
555
+ // with the $-delimited LaTeX as the visible (escaped) text content.
556
+ if ((tagName === 'div' || tagName === 'span') && node.attributes?.['data-math'] !== undefined) {
557
+ const mathMode = node.attributes['data-math'] === 'block' ? 'block' : 'inline';
558
+ const rawText = node.children.map(c => c.text || '').join('')
559
+ .replace(/&nbsp;/g, ' ')
560
+ .replace(/&lt;/g, '<')
561
+ .replace(/&gt;/g, '>')
562
+ .replace(/&amp;/g, '&')
563
+ .replace(/&quot;/g, '"')
564
+ .replace(/&#39;/g, '\'');
565
+ const delimiter = mathMode === 'block' ? '$$' : '$';
566
+ const latex = rawText.startsWith(delimiter) && rawText.endsWith(delimiter)
567
+ ? rawText.slice(delimiter.length, -delimiter.length)
568
+ : rawText;
569
+ return {
570
+ type: 'code',
571
+ text: latex,
572
+ metadata: { math: mathMode }
573
+ };
574
+ }
575
+ // Admonition: inscript-editor's Admonition node renders
576
+ // <div class="admonition admonition-note" data-type="note">…children…</div>.
577
+ if (tagName === 'div' && (node.attributes?.class || '').split(/\s+/).includes('admonition')) {
578
+ const admonitionTypeAttr = node.attributes?.['data-type'];
579
+ const admonitionType = ['note', 'tip', 'important', 'warning', 'caution'].includes(admonitionTypeAttr)
580
+ ? admonitionTypeAttr
581
+ : 'note';
582
+ const admonitionNode = {
583
+ type: 'admonition',
584
+ metadata: { admonitionType },
585
+ children: parseChildren(node, newFormatting, listContext)
586
+ };
587
+ if (config.includeRawContent)
588
+ admonitionNode.rawContent = '<div class="admonition">...</div>';
589
+ return admonitionNode;
590
+ }
277
591
  // Skip structural containers produced by HtmlGenerator to avoid deep AST nesting
278
592
  if (tagName === 'div' && (node.attributes?.class === 'container' ||
279
593
  node.attributes?.class === 'spreadsheet-container' ||
@@ -296,7 +610,7 @@ const parseHtml = async (buffer, config) => {
296
610
  if (tagName === 'p' || tagName === 'div') {
297
611
  const children = parseChildren(node, newFormatting, listContext);
298
612
  // If it's a div and contains block elements, return children directly
299
- const hasBlockElements = children.some(c => ['paragraph', 'table', 'heading', 'list', 'image', 'chart', 'code'].includes(c.type));
613
+ const hasBlockElements = children.some(c => ['paragraph', 'table', 'heading', 'list', 'image', 'chart', 'code', 'embed', 'admonition', 'definitionList'].includes(c.type));
300
614
  if (tagName === 'div' && hasBlockElements) {
301
615
  return children;
302
616
  }
@@ -313,7 +627,8 @@ const parseHtml = async (buffer, config) => {
313
627
  const pNode = {
314
628
  type: 'paragraph',
315
629
  metadata: { alignment: newFormatting.alignment, anchorIds: anchorIds.length > 0 ? anchorIds : undefined },
316
- children: flattenedChildren
630
+ children: flattenedChildren,
631
+ htmlAttributes: collectHtmlAttributes(node, ['align'])
317
632
  };
318
633
  if (config.includeRawContent) {
319
634
  // Note: Since this is a manual parser without locators, we can't easily get the original source slice.
@@ -326,17 +641,58 @@ const parseHtml = async (buffer, config) => {
326
641
  const hNode = {
327
642
  type: 'heading',
328
643
  metadata: { level, alignment: newFormatting.alignment, anchorIds: anchorIds.length > 0 ? anchorIds : undefined },
329
- children: parseChildren(node, newFormatting, listContext)
644
+ children: parseChildren(node, newFormatting, listContext),
645
+ htmlAttributes: collectHtmlAttributes(node, ['align'])
330
646
  };
331
647
  return hNode;
332
648
  }
649
+ if (tagName === 'dl') {
650
+ return {
651
+ type: 'definitionList',
652
+ children: parseChildren(node, newFormatting, listContext)
653
+ };
654
+ }
655
+ if (tagName === 'dt') {
656
+ return {
657
+ type: 'definitionTerm',
658
+ children: parseChildren(node, newFormatting, listContext)
659
+ };
660
+ }
661
+ if (tagName === 'dd') {
662
+ return {
663
+ type: 'definitionDescription',
664
+ children: parseChildren(node, newFormatting, listContext)
665
+ };
666
+ }
667
+ if (tagName === 'abbr') {
668
+ const title = node.attributes?.title;
669
+ const children = parseChildren(node, newFormatting, listContext);
670
+ if (title) {
671
+ children.forEach(c => {
672
+ if (c.type === 'text') {
673
+ c.metadata = { ...c.metadata, abbreviationTitle: title };
674
+ }
675
+ });
676
+ }
677
+ return children;
678
+ }
679
+ if (tagName === 'cite' && node.attributes?.['data-citation-key'] !== undefined) {
680
+ const citationKey = node.attributes['data-citation-key'];
681
+ return {
682
+ type: 'text',
683
+ text: citationKey,
684
+ formatting: Object.keys(newFormatting).length > 0 ? { ...newFormatting } : undefined,
685
+ metadata: { citationKey }
686
+ };
687
+ }
333
688
  if (tagName === 'ul' || tagName === 'ol') {
334
689
  const isNewTopLevel = !listContext;
335
690
  const newListContext = {
336
691
  listId: isNewTopLevel ? `html-list-${htmlListIdCounter++}` : listContext.listId,
337
692
  type: tagName === 'ol' ? 'ordered' : 'unordered',
338
693
  level: isNewTopLevel ? 0 : listContext.level + 1,
339
- counters: isNewTopLevel ? {} : { ...listContext.counters } // Clone to avoid side effects on parent levels
694
+ counters: isNewTopLevel ? {} : { ...listContext.counters }, // Clone to avoid side effects on parent levels
695
+ isTask: node.attributes?.['data-type'] === 'taskList'
340
696
  };
341
697
  // Initialize counter for this level
342
698
  if (tagName === 'ol' && node.attributes?.start) {
@@ -362,6 +718,13 @@ const parseHtml = async (buffer, config) => {
362
718
  const children = parseChildren(node, newFormatting, listContext);
363
719
  const nestedLists = children.filter(c => c.type === 'list');
364
720
  const selfChildren = children.filter(c => c.type !== 'list');
721
+ let isTask;
722
+ let checked;
723
+ if (listContext?.isTask) {
724
+ isTask = true;
725
+ const dataChecked = node.attributes?.['data-checked'];
726
+ checked = dataChecked !== undefined ? dataChecked === 'true' : (findNestedCheckboxChecked(node) ?? false);
727
+ }
365
728
  const selfNode = {
366
729
  type: 'list',
367
730
  text: selfChildren.map(c => c.text || '').join(''),
@@ -371,17 +734,23 @@ const parseHtml = async (buffer, config) => {
371
734
  alignment: newFormatting.alignment || 'left',
372
735
  listId: listContext?.listId || 'html-list-none',
373
736
  itemIndex: (listContext?.counters[listContext.level] ?? 1) - 1,
374
- anchorIds: anchorIds.length > 0 ? anchorIds : undefined
737
+ anchorIds: anchorIds.length > 0 ? anchorIds : undefined,
738
+ isTask,
739
+ checked
375
740
  },
376
741
  children: selfChildren
377
742
  };
378
743
  return [selfNode, ...nestedLists];
379
744
  }
380
745
  if (tagName === 'table') {
746
+ // CustomTable (inscript-editor) renders data-align on the <table> itself.
747
+ const tableAlignAttr = node.attributes?.['data-align'];
748
+ const tableAlign = ['left', 'center', 'right'].includes(tableAlignAttr) ? tableAlignAttr : undefined;
381
749
  const tableNode = {
382
750
  type: 'table',
383
- metadata: { anchorIds: anchorIds.length > 0 ? anchorIds : undefined },
384
- children: parseChildren(node, newFormatting, listContext)
751
+ metadata: { anchorIds: anchorIds.length > 0 ? anchorIds : undefined, align: tableAlign },
752
+ children: parseChildren(node, newFormatting, listContext),
753
+ htmlAttributes: collectHtmlAttributes(node, ['data-align', 'align'])
385
754
  };
386
755
  if (config.includeRawContent) {
387
756
  tableNode.rawContent = '<table>...</table>';
@@ -391,7 +760,8 @@ const parseHtml = async (buffer, config) => {
391
760
  if (tagName === 'tr') {
392
761
  const rowNode = {
393
762
  type: 'row',
394
- children: parseChildren(node, newFormatting, listContext)
763
+ children: parseChildren(node, newFormatting, listContext),
764
+ htmlAttributes: collectHtmlAttributes(node, [])
395
765
  };
396
766
  if (config.includeRawContent) {
397
767
  rowNode.rawContent = '<tr>...</tr>';
@@ -399,9 +769,20 @@ const parseHtml = async (buffer, config) => {
399
769
  return rowNode;
400
770
  }
401
771
  if (tagName === 'td' || tagName === 'th') {
772
+ // Merged cells: mirrors the colspan/rowspan reading already done in
773
+ // MarkdownParser's inline HTML-table handler.
774
+ const colSpanAttr = node.attributes?.colspan;
775
+ const rowSpanAttr = node.attributes?.rowspan;
776
+ const colSpan = colSpanAttr ? parseInt(colSpanAttr, 10) : undefined;
777
+ const rowSpan = rowSpanAttr ? parseInt(rowSpanAttr, 10) : undefined;
402
778
  const cellNode = {
403
779
  type: 'cell',
404
- children: parseChildren(node, newFormatting, listContext)
780
+ metadata: {
781
+ colSpan: colSpan && !isNaN(colSpan) ? colSpan : undefined,
782
+ rowSpan: rowSpan && !isNaN(rowSpan) ? rowSpan : undefined
783
+ },
784
+ children: parseChildren(node, newFormatting, listContext),
785
+ htmlAttributes: collectHtmlAttributes(node, ['colspan', 'rowspan', 'align'])
405
786
  };
406
787
  if (config.includeRawContent) {
407
788
  cellNode.rawContent = '<td>...</td>';
@@ -411,6 +792,30 @@ const parseHtml = async (buffer, config) => {
411
792
  if (tagName === 'img') {
412
793
  const src = node.attributes?.src;
413
794
  const alt = node.attributes?.alt;
795
+ // CustomImage (inscript-editor) renders data-width/data-align, falling back to
796
+ // parsing the inline style for consumers that only emit the CSS.
797
+ const imgDecls = parseStyleDeclarations(node.attributes?.style || '');
798
+ // Exact lookup, so `max-width: 100%` - the standard responsive-image style, and by
799
+ // far the most common inline style on an <img> - is no longer read as a declared
800
+ // width. It constrains the rendered size; it is not an author-specified width.
801
+ const width = node.attributes?.['data-width'] || getDeclaration(imgDecls, 'width');
802
+ // Alignment is inferred from which auto margin is present. Comparing the parsed
803
+ // value rather than substring-matching "margin-left: 0" stops `margin-left: 0.5rem`
804
+ // from being read as left-aligned, and lets the `margin: 0 auto` centering
805
+ // shorthand be recognised at all.
806
+ const marginLeft = getDeclaration(imgDecls, 'margin-left');
807
+ const marginRight = getDeclaration(imgDecls, 'margin-right');
808
+ const marginShorthand = getDeclaration(imgDecls, 'margin');
809
+ const isZero = (v) => v !== undefined && /^0(?:[a-z%]*)$/.test(v);
810
+ const shorthandParts = marginShorthand ? marginShorthand.split(/\s+/) : [];
811
+ const shorthandCentres = shorthandParts.length > 1
812
+ && shorthandParts[shorthandParts.length - 1] === 'auto'
813
+ && shorthandParts[1] === 'auto';
814
+ const alignAttr = node.attributes?.['data-align']
815
+ ?? (shorthandCentres ? 'center'
816
+ : (isZero(marginLeft) && !isZero(marginRight) ? 'left'
817
+ : (isZero(marginRight) && !isZero(marginLeft) ? 'right' : undefined)));
818
+ const align = ['left', 'center', 'right'].includes(alignAttr) ? alignAttr : undefined;
414
819
  let imageNode;
415
820
  if (src?.startsWith('data:')) {
416
821
  const match = src.match(/^data:([^;]+);base64,(.*)$/);
@@ -429,7 +834,9 @@ const parseHtml = async (buffer, config) => {
429
834
  type: 'image',
430
835
  metadata: {
431
836
  attachmentName: name,
432
- altText: alt
837
+ altText: alt,
838
+ width,
839
+ align
433
840
  }
434
841
  };
435
842
  }
@@ -438,7 +845,9 @@ const parseHtml = async (buffer, config) => {
438
845
  type: 'image',
439
846
  metadata: {
440
847
  url: src,
441
- altText: alt
848
+ altText: alt,
849
+ width,
850
+ align
442
851
  }
443
852
  };
444
853
  }
@@ -449,7 +858,9 @@ const parseHtml = async (buffer, config) => {
449
858
  metadata: {
450
859
  url: src,
451
860
  altText: alt,
452
- anchorIds: anchorIds.length > 0 ? anchorIds : undefined
861
+ anchorIds: anchorIds.length > 0 ? anchorIds : undefined,
862
+ width,
863
+ align
453
864
  }
454
865
  };
455
866
  }
@@ -460,8 +871,16 @@ const parseHtml = async (buffer, config) => {
460
871
  }
461
872
  if (tagName === 'a') {
462
873
  const href = node.attributes?.href;
874
+ const wikilinkPage = node.attributes?.['data-wikilink-page'];
463
875
  const children = parseChildren(node, newFormatting, listContext);
464
- if (href) {
876
+ if (wikilinkPage !== undefined) {
877
+ children.forEach(c => {
878
+ if (c.type === 'text') {
879
+ c.metadata = { ...c.metadata, link: wikilinkPage, linkType: 'internal', wikilink: true };
880
+ }
881
+ });
882
+ }
883
+ else if (href) {
465
884
  const linkType = href.startsWith('#') ? 'internal' : 'external';
466
885
  children.forEach(c => {
467
886
  if (c.type === 'text') {
@@ -509,6 +928,42 @@ const parseHtml = async (buffer, config) => {
509
928
  }
510
929
  return null;
511
930
  };
931
+ // Extract <section data-footnotes> up front so its definitions are available to
932
+ // <sup data-footnote-ref> references encountered anywhere earlier in the body.
933
+ const findFootnotesSection = (n) => {
934
+ if (n.tagName === 'section' && n.attributes?.['data-footnotes'] !== undefined)
935
+ return n;
936
+ for (const child of n.children) {
937
+ const found = findFootnotesSection(child);
938
+ if (found)
939
+ return found;
940
+ }
941
+ return undefined;
942
+ };
943
+ const footnotesSectionNode = findFootnotesSection(body);
944
+ if (footnotesSectionNode) {
945
+ for (const item of footnotesSectionNode.children) {
946
+ if (item.type !== 'element')
947
+ continue;
948
+ const key = item.attributes?.['data-footnote-id'];
949
+ if (!key)
950
+ continue;
951
+ // Strip the generated back-reference link ("↩") - it's round-trip plumbing,
952
+ // not part of the footnote's actual content.
953
+ const filteredChildren = item.children.filter(c => !(c.tagName === 'a' && (c.attributes?.href || '').startsWith('#footnote-ref-')));
954
+ const contentNodes = [];
955
+ for (const child of filteredChildren) {
956
+ const parsed = parseNode(child);
957
+ if (parsed) {
958
+ if (Array.isArray(parsed))
959
+ contentNodes.push(...parsed);
960
+ else
961
+ contentNodes.push(parsed);
962
+ }
963
+ }
964
+ footnoteDefinitions.set(key, contentNodes);
965
+ }
966
+ }
512
967
  for (const child of body.children) {
513
968
  const parsed = parseNode(child);
514
969
  if (parsed) {
@@ -539,8 +994,12 @@ const parseHtml = async (buffer, config) => {
539
994
  return node.text || '';
540
995
  if (node.type === 'break')
541
996
  return '\n';
997
+ // Childless nodes still carry meaningful text - fall back to it instead of
998
+ // silently vanishing from plain-text/RAG-chunk output.
999
+ if (node.type === 'embed')
1000
+ return node.metadata?.url || '';
542
1001
  if (node.children) {
543
- const isBlock = ['table', 'row', 'list', 'sheet', 'slide'].includes(node.type);
1002
+ const isBlock = ['table', 'row', 'list', 'sheet', 'slide', 'admonition', 'definitionList'].includes(node.type);
544
1003
  return node.children.map(getText).join(isBlock ? config.newlineDelimiter : '');
545
1004
  }
546
1005
  return '';