officeparser 7.2.3 → 7.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/README.md +277 -17
  2. package/dist/OfficeConverter.d.ts +1 -1
  3. package/dist/OfficeConverter.js +3 -0
  4. package/dist/OfficeGenerator.js +4 -0
  5. package/dist/OfficeParser.d.ts +2 -0
  6. package/dist/OfficeParser.js +6 -0
  7. package/dist/cli.d.ts +1 -1
  8. package/dist/cli.js +5 -2
  9. package/dist/defaults.js +12 -0
  10. package/dist/generators/BaseGenerator.d.ts +34 -1
  11. package/dist/generators/BaseGenerator.js +98 -0
  12. package/dist/generators/CsvGenerator.d.ts +9 -1
  13. package/dist/generators/CsvGenerator.js +28 -16
  14. package/dist/generators/EpubGenerator.d.ts +43 -0
  15. package/dist/generators/EpubGenerator.js +312 -0
  16. package/dist/generators/HtmlGenerator.d.ts +12 -0
  17. package/dist/generators/HtmlGenerator.js +378 -61
  18. package/dist/generators/MarkdownGenerator.d.ts +28 -5
  19. package/dist/generators/MarkdownGenerator.js +432 -51
  20. package/dist/generators/PdfGenerator.js +32 -0
  21. package/dist/generators/RtfGenerator.js +47 -22
  22. package/dist/generators/TextGenerator.js +98 -11
  23. package/dist/index.d.ts +1 -0
  24. package/dist/index.js +1 -0
  25. package/dist/officeparser.browser.d.ts +427 -20
  26. package/dist/officeparser.browser.iife.js +338 -206
  27. package/dist/officeparser.browser.mjs +346 -214
  28. package/dist/officeparser.browser.slim.d.ts +427 -20
  29. package/dist/officeparser.browser.slim.iife.js +346 -214
  30. package/dist/officeparser.browser.slim.mjs +346 -214
  31. package/dist/parsers/EpubParser.d.ts +8 -0
  32. package/dist/parsers/EpubParser.js +217 -0
  33. package/dist/parsers/ExcelParser.js +2 -0
  34. package/dist/parsers/HtmlParser.js +507 -48
  35. package/dist/parsers/MarkdownParser.js +704 -92
  36. package/dist/parsers/OpenOfficeParser.js +128 -20
  37. package/dist/parsers/PdfParser.js +4 -1
  38. package/dist/parsers/PowerPointParser.js +1 -0
  39. package/dist/parsers/WordParser.js +1 -0
  40. package/dist/sbom.cdx.json +1695 -0
  41. package/dist/types.d.ts +427 -20
  42. package/dist/types.js +8 -0
  43. package/dist/utils/configUtils.js +53 -4
  44. package/dist/utils/errorUtils.js +7 -3
  45. package/dist/utils/sanitize.d.ts +139 -0
  46. package/dist/utils/sanitize.js +318 -0
  47. package/dist/utils/xmlUtils.js +2 -2
  48. package/dist/utils/zipUtils.js +76 -26
  49. package/package.json +16 -12
@@ -3,6 +3,53 @@ Object.defineProperty(exports, "__esModule", { value: true });
3
3
  exports.parseMarkdown = void 0;
4
4
  const astUtils_js_1 = require("../utils/astUtils.js");
5
5
  const errorUtils_js_1 = require("../utils/errorUtils.js");
6
+ // Sentinel node type for a standalone bookmark-anchor block (e.g. `<a id="x"></a>` on its
7
+ // own line). A post-parse pass folds these into the following node's anchorIds so they
8
+ // round-trip as real anchors rather than being escaped to visible text on regeneration.
9
+ const ANCHOR_PLACEHOLDER = '__anchorPlaceholder__';
10
+ /**
11
+ * Splits the inner content of a YAML flow array (`a, "b, c", d`) on top-level commas,
12
+ * ignoring commas inside single- or double-quoted items.
13
+ */
14
+ const splitFlowArrayItems = (inner) => {
15
+ const items = [];
16
+ let current = '';
17
+ let quote = null;
18
+ for (const ch of inner) {
19
+ if (quote) {
20
+ current += ch;
21
+ if (ch === quote)
22
+ quote = null;
23
+ }
24
+ else if (ch === '"' || ch === '\'') {
25
+ quote = ch;
26
+ current += ch;
27
+ }
28
+ else if (ch === ',') {
29
+ items.push(current.trim());
30
+ current = '';
31
+ }
32
+ else {
33
+ current += ch;
34
+ }
35
+ }
36
+ if (current.trim() !== '')
37
+ items.push(current.trim());
38
+ return items;
39
+ };
40
+ /**
41
+ * Maps every accepted-on-import admonition type spelling (GitHub's five plus GLFM's
42
+ * `danger`) to the canonical AdmonitionMetadata type. Per MARKDOWN_DIALECT.md's
43
+ * Decisions, `danger` folds into `caution` - there is no separate danger type.
44
+ */
45
+ const ADMONITION_TYPE_MAP = {
46
+ note: 'note',
47
+ tip: 'tip',
48
+ important: 'important',
49
+ warning: 'warning',
50
+ caution: 'caution',
51
+ danger: 'caution'
52
+ };
6
53
  const parseMarkdown = async (buffer, config) => {
7
54
  // Honour cancellation requests before the line-by-line Markdown scanning loop begins.
8
55
  // Markdown parsing is entirely synchronous and CPU-bound, so failing fast avoids
@@ -26,9 +73,23 @@ const parseMarkdown = async (buffer, config) => {
26
73
  const match = line.match(/^([^:]+):\s*(.*)$/);
27
74
  if (match) {
28
75
  const key = match[1].trim();
29
- let val = match[2].trim().replace(/^"(.*)"$/, '$1');
76
+ const rawVal = match[2].trim();
77
+ const val = rawVal.replace(/^"(.*)"$/, '$1');
30
78
  let parsedVal = val;
31
- if (val === 'true')
79
+ if (rawVal.startsWith('[') && rawVal.endsWith(']')) {
80
+ // Flow-array (`tags: [a, b]`) or JSON-array (`tags: ["a","b"]`) value -
81
+ // parse into a real array instead of storing the literal bracket string,
82
+ // so it round-trips symmetrically with MarkdownGenerator's frontmatter output.
83
+ try {
84
+ const jsonParsed = JSON.parse(rawVal);
85
+ parsedVal = Array.isArray(jsonParsed) ? jsonParsed : val;
86
+ }
87
+ catch {
88
+ const inner = rawVal.slice(1, -1).trim();
89
+ parsedVal = inner === '' ? [] : splitFlowArrayItems(inner).map(item => item.replace(/^['"](.*)['"]$/, '$1'));
90
+ }
91
+ }
92
+ else if (val === 'true')
32
93
  parsedVal = true;
33
94
  else if (val === 'false')
34
95
  parsedVal = false;
@@ -56,88 +117,381 @@ const parseMarkdown = async (buffer, config) => {
56
117
  metadata.nativeProperties = nativeProps;
57
118
  }
58
119
  }
59
- // Extract code blocks first to protect their contents
120
+ // Strip MDX/JSX component tags (parse-only - we never author MDX). Components are
121
+ // distinguished from plain HTML by an uppercase-leading tag name, matching React/MDX
122
+ // convention. Self-closing components are removed entirely; paired components keep
123
+ // their inner Markdown content. Iterate to a fixed point so nested components (of
124
+ // different names) are all unwrapped, not just the outermost one.
125
+ // Cap the passes: each iteration unwraps one nesting level, so a pathologically
126
+ // deep `<A><A>...</A></A>` input would otherwise be O(depth * n). Real documents
127
+ // nest only a handful of levels; anything past the cap is left as-is.
128
+ let previousTextStr;
129
+ let mdxPasses = 0;
130
+ const MAX_MDX_PASSES = 100;
131
+ do {
132
+ previousTextStr = textStr;
133
+ textStr = textStr.replace(/<[A-Z][A-Za-z0-9]*(?:\s+[^>]*?)?\/>/g, '');
134
+ textStr = textStr.replace(/<([A-Z][A-Za-z0-9]*)(?:\s+[^>]*?)?>([\s\S]*?)<\/\1>/g, (_m, _name, inner) => inner);
135
+ } while (textStr !== previousTextStr && ++mdxPasses < MAX_MDX_PASSES);
136
+ // Extract code blocks first to protect their contents. Accepts both backtick and
137
+ // tilde fences (CommonMark's two fence characters); the backreference on the fence
138
+ // run means a `~~~`-fenced block isn't closed early by a stray ``` inside it, and
139
+ // vice versa.
60
140
  const codeBlocks = [];
61
- textStr = textStr.replace(/^```(\w*)\n([\s\S]*?)\n```/gm, (match, lang, code) => {
141
+ textStr = textStr.replace(/^(`{3,}|~{3,})(\w*)\n([\s\S]*?)\n\1$/gm, (match, _fence, lang, code) => {
62
142
  const id = `__CODE_BLOCK_${codeBlocks.length}__`;
63
143
  codeBlocks.push(JSON.stringify({ lang, code }));
64
144
  return `\n\n${id}\n\n`;
65
145
  });
146
+ // Extract block math ($$\n...\n$$) before block splitting, mirroring the code-block
147
+ // pre-pass above - its body may contain blank lines that would otherwise fragment it.
148
+ // Inline math ($...$) is handled directly in parseInline below.
149
+ const mathBlocks = [];
150
+ textStr = textStr.replace(/^\$\$\n([\s\S]*?)\n\$\$$/gm, (_match, latex) => {
151
+ const id = `__MATH_BLOCK_${mathBlocks.length}__`;
152
+ mathBlocks.push(latex);
153
+ return `\n\n${id}\n\n`;
154
+ });
155
+ // Extract GLFM-style fenced-div admonitions (`:::note ... :::`) before block splitting,
156
+ // since their body may itself contain blank lines that would otherwise fragment them.
157
+ // The `> [!NOTE]` GitHub form doesn't need this - it's detected inline in the blockquote
158
+ // branch below, since a `>`-prefixed block never contains a real blank line.
159
+ const admonitionBlocks = [];
160
+ textStr = textStr.replace(/^:::(\w+)[ \t]*\n([\s\S]*?)\n:::[ \t]*$/gm, (match, type, body) => {
161
+ const admonitionType = ADMONITION_TYPE_MAP[type.toLowerCase()];
162
+ if (!admonitionType)
163
+ return match; // Unrecognised type - leave as literal text.
164
+ const id = `__ADMONITION_${admonitionBlocks.length}__`;
165
+ admonitionBlocks.push(JSON.stringify({ admonitionType, body }));
166
+ return `\n\n${id}\n\n`;
167
+ });
168
+ // Extract footnote definitions (`[^id]: text`) before block splitting, since
169
+ // definitions conventionally live at the end of the document, after every place
170
+ // they're referenced - inline parsing below needs the full map upfront. v1 only
171
+ // supports single-line definitions (MultiMarkdown/Pandoc/GLFM's common baseline).
172
+ const footnoteDefinitions = new Map();
173
+ textStr = textStr.replace(/^\[\^([^\]]+)\]:[ \t]*(.*)$/gm, (_match, id, definition) => {
174
+ footnoteDefinitions.set(id, definition.trim());
175
+ return '';
176
+ });
177
+ // Extract Markdown Extra abbreviation definitions (`*[HTML]: Hypertext Markup Language`)
178
+ // before block splitting, for the same reason as footnotes: they conventionally live
179
+ // at the end of the document.
180
+ const abbreviationDefinitions = new Map();
181
+ textStr = textStr.replace(/^\*\[([^\]]+)\]:[ \t]*(.*)$/gm, (_match, abbr, definition) => {
182
+ abbreviationDefinitions.set(abbr, definition.trim());
183
+ return '';
184
+ });
185
+ // Extract link/image reference definitions (`[ref]: /url "title"`) before block
186
+ // splitting, for the same reason as footnotes/abbreviations: they conventionally
187
+ // live at the end of the document, after every place they're referenced. Keyed by
188
+ // trimmed/lowercased label, matching CommonMark's case-insensitive reference matching.
189
+ const linkDefinitions = new Map();
190
+ textStr = textStr.replace(/^\[([^\]]+)\]:[ \t]*(\S+)(?:[ \t]+"([^"]*)")?[ \t]*$/gm, (_match, label, url, title) => {
191
+ linkDefinitions.set(label.trim().toLowerCase(), { url, title });
192
+ return '';
193
+ });
194
+ // Parses a Pandoc-style attribute list body (the part inside `{...}`), e.g.
195
+ // `width=50% .centered` or `align=right`. Per MARKDOWN_DIALECT.md §15's Decisions,
196
+ // the vocabulary matches ImageMetadata/TableMetadata's own width/align fields;
197
+ // several class-name spellings are accepted on import for compatibility with
198
+ // hand-written content, but the generator only ever emits canonical `align=value`.
199
+ const parseAttributeList = (attrStr) => {
200
+ const result = {};
201
+ for (const token of attrStr.trim().split(/\s+/).filter(Boolean)) {
202
+ const kv = token.match(/^([a-zA-Z-]+)=(.+)$/);
203
+ if (kv) {
204
+ if (kv[1] === 'width')
205
+ result.width = kv[2];
206
+ else if (kv[1] === 'align' && ['left', 'center', 'right'].includes(kv[2]))
207
+ result.align = kv[2];
208
+ }
209
+ else if (token.startsWith('.')) {
210
+ const cls = token.slice(1).toLowerCase();
211
+ if (cls === 'left' || cls === 'align-left')
212
+ result.align = 'left';
213
+ else if (cls === 'center' || cls === 'centered' || cls === 'align-center')
214
+ result.align = 'center';
215
+ else if (cls === 'right' || cls === 'align-right')
216
+ result.align = 'right';
217
+ }
218
+ }
219
+ return result;
220
+ };
66
221
  const parseInline = (text, currentFormatting = {}) => {
67
222
  const nodes = [];
68
- // Regex matches: 1=!, 2=alt, 3=url | 4=bold | 5=italic | 6=strike | 7=code | 8=underline | 9=subscript | 10=superscript
69
- const regex = /(!?)\[(.*?)\]\((.*?)\)|\*\*(.+?)\*\*|\*(.+?)\*|~~(.+?)~~|`(.+?)`|<u>(.+?)<\/u>|<sub>(.+?)<\/sub>|<sup>(.+?)<\/sup>/g;
223
+ const plainText = (t) => ({ type: 'text', text: t, formatting: Object.keys(currentFormatting).length > 0 ? { ...currentFormatting } : undefined });
224
+ // Builds the same image/link node shape regardless of whether the URL came from
225
+ // an inline `(url)` or a resolved reference definition - shared by the inline
226
+ // image/link branch and the two reference-style branches below.
227
+ const buildLinkOrImageNodes = (isImage, altText, url, attrsStr) => {
228
+ if (isImage) {
229
+ // Pandoc-style attribute list immediately after an image, e.g. {width=50% .centered}
230
+ const attrs = attrsStr !== undefined ? parseAttributeList(attrsStr) : undefined;
231
+ if (url.startsWith('data:')) {
232
+ const dataMatch = url.match(/^data:([^;]+);base64,(.*)$/);
233
+ if (dataMatch && config.extractAttachments) {
234
+ const mimeType = dataMatch[1];
235
+ const data = dataMatch[2];
236
+ const name = `image_${attachments.length + 1}.${mimeType.split('/')[1]}`;
237
+ attachments.push({
238
+ type: 'image',
239
+ mimeType,
240
+ data,
241
+ name,
242
+ extension: mimeType.split('/')[1]
243
+ });
244
+ return [{ type: 'image', metadata: { attachmentName: name, altText, ...attrs } }];
245
+ }
246
+ }
247
+ return [{ type: 'image', metadata: { url, altText, ...attrs } }];
248
+ }
249
+ const linkNodes = parseInline(altText, currentFormatting);
250
+ linkNodes.forEach(n => {
251
+ if (n.type === 'text') {
252
+ n.metadata = { link: url, linkType: 'external' };
253
+ }
254
+ });
255
+ return linkNodes;
256
+ };
257
+ // Regex matches (named groups): esc=escaped punctuation char | imgBang/imgAlt/imgUrl/imgAttrs=inline
258
+ // image or link | boldStar/boldUnderscore=bold | italicStar/italicUnderscore=italic | strike=strikethrough |
259
+ // codeFence/codeContent=inline code (backreferenced fence run, so a shorter embedded backtick run
260
+ // doesn't close the span early) | underline/subscript/superscript=HTML tag formatting |
261
+ // footnoteId | citationKey | wikiPage/wikiAlias | refBang/refText/refId=explicit or collapsed
262
+ // reference link/image `[text][ref]`/`[text][]` | shortBang/shortText=shortcut reference `[text]`
263
+ // (deliberately the most generic bracket pattern, so it must stay last among `[`-starting
264
+ // alternatives) | autolinkUrl=`<url>` autolink | mathInline.
265
+ //
266
+ // Named groups (rather than positional match[N] indices) mean adding a new alternative never
267
+ // requires renumbering every existing dispatch arm.
268
+ //
269
+ // Escape must be listed first since only a literal backslash can start that alternative, so it
270
+ // never shadows another branch; but a code span's match consumes its whole span atomically (the
271
+ // exec loop's lastIndex jumps past the entire matched span), so a backslash *inside* a code span
272
+ // is never independently offered to the escape branch regardless of listing order - CommonMark's
273
+ // "backslashes are not special inside code spans" rule holds by construction, not extra logic.
274
+ //
275
+ // Underscore emphasis has no CommonMark flanking-delimiter-run detection, so an intraword
276
+ // underscore (e.g. "foo_bar_baz") will incorrectly italicize - an accepted, documented
277
+ // simplification, not something this pass attempts to fix.
278
+ //
279
+ // Inline math requires no whitespace right after the opening $ or right before the
280
+ // closing $, the common heuristic (matching Pandoc/KaTeX) for avoiding false
281
+ // positives on currency like "$5 and $10".
282
+ const regex = /\\(?<esc>[!-\/:-@\[-`{-~])|(?<imgBang>!?)\[(?<imgAlt>.*?)\]\((?<imgUrl>.*?)\)(?:\{(?<imgAttrs>[^}]*)\})?|\*\*(?<boldStar>.+?)\*\*|__(?<boldUnderscore>.+?)__|\*(?<italicStar>.+?)\*|_(?<italicUnderscore>.+?)_|~~(?<strike>.+?)~~|(?<codeFence>`+)(?<codeContent>(?:(?!\k<codeFence>)[\s\S])+?)\k<codeFence>(?!`)|<u>(?<underline>.+?)<\/u>|<sub>(?<subscript>.+?)<\/sub>|<sup>(?<superscript>.+?)<\/sup>|\[\^(?<footnoteId>[^\]]+)\]|\[@(?<citationKey>[a-zA-Z0-9_:.-]+)\]|\[\[(?<wikiPage>[^\]|]+)(?:\|(?<wikiAlias>[^\]]+))?\]\]|(?<refBang>!?)\[(?<refText>[^\]]*)\]\[(?<refId>[^\]]*)\]|(?<shortBang>!?)\[(?<shortText>[^\]]+)\]|<(?<autolinkUrl>(?:https?|mailto):[^\s<>]+)>|\$(?!\s)(?<mathInline>[^$\n]+?)(?<!\s)\$/g;
70
283
  let lastIndex = 0;
71
284
  let match;
72
285
  while ((match = regex.exec(text)) !== null) {
73
286
  if (match.index > lastIndex) {
74
- nodes.push({ type: 'text', text: text.substring(lastIndex, match.index), formatting: Object.keys(currentFormatting).length > 0 ? { ...currentFormatting } : undefined });
75
- }
76
- if (match[2] !== undefined) { // Image or Link
77
- const isImage = match[1] === '!';
78
- const altText = match[2];
79
- const url = match[3];
80
- if (isImage) {
81
- if (url.startsWith('data:')) {
82
- const dataMatch = url.match(/^data:([^;]+);base64,(.*)$/);
83
- if (dataMatch && config.extractAttachments) {
84
- const mimeType = dataMatch[1];
85
- const data = dataMatch[2];
86
- const name = `image_${attachments.length + 1}.${mimeType.split('/')[1]}`;
87
- attachments.push({
88
- type: 'image',
89
- mimeType,
90
- data,
91
- name,
92
- extension: mimeType.split('/')[1]
93
- });
94
- nodes.push({ type: 'image', metadata: { attachmentName: name, altText } });
95
- }
96
- else {
97
- nodes.push({ type: 'image', metadata: { url, altText } });
98
- }
99
- }
100
- else {
101
- nodes.push({ type: 'image', metadata: { url, altText } });
102
- }
287
+ nodes.push(plainText(text.substring(lastIndex, match.index)));
288
+ }
289
+ const g = match.groups;
290
+ if (g.esc !== undefined) { // Backslash-escaped punctuation
291
+ nodes.push(plainText(g.esc));
292
+ }
293
+ else if (g.imgAlt !== undefined) { // Image or Link
294
+ nodes.push(...buildLinkOrImageNodes(g.imgBang === '!', g.imgAlt, g.imgUrl, g.imgAttrs));
295
+ }
296
+ else if (g.boldStar !== undefined) { // Bold (**)
297
+ nodes.push(...parseInline(g.boldStar, { ...currentFormatting, bold: true }));
298
+ }
299
+ else if (g.boldUnderscore !== undefined) { // Bold (__)
300
+ nodes.push(...parseInline(g.boldUnderscore, { ...currentFormatting, bold: true }));
301
+ }
302
+ else if (g.italicStar !== undefined) { // Italic (*)
303
+ nodes.push(...parseInline(g.italicStar, { ...currentFormatting, italic: true }));
304
+ }
305
+ else if (g.italicUnderscore !== undefined) { // Italic (_)
306
+ nodes.push(...parseInline(g.italicUnderscore, { ...currentFormatting, italic: true }));
307
+ }
308
+ else if (g.strike !== undefined) { // Strikethrough
309
+ nodes.push(...parseInline(g.strike, { ...currentFormatting, strikethrough: true }));
310
+ }
311
+ else if (g.codeContent !== undefined) { // Inline code (any matching backtick-run length)
312
+ nodes.push({ type: 'text', text: g.codeContent, formatting: { ...currentFormatting, font: 'monospace' } });
313
+ }
314
+ else if (g.underline !== undefined) { // Underline
315
+ nodes.push(...parseInline(g.underline, { ...currentFormatting, underline: true }));
316
+ }
317
+ else if (g.subscript !== undefined) { // Subscript
318
+ nodes.push(...parseInline(g.subscript, { ...currentFormatting, subscript: true }));
319
+ }
320
+ else if (g.superscript !== undefined) { // Superscript
321
+ nodes.push(...parseInline(g.superscript, { ...currentFormatting, superscript: true }));
322
+ }
323
+ else if (g.footnoteId !== undefined) { // Footnote reference
324
+ const noteId = g.footnoteId;
325
+ const definition = footnoteDefinitions.get(noteId);
326
+ const noteChildren = definition !== undefined ? parseInline(definition) : [];
327
+ const noteNode = {
328
+ type: 'note',
329
+ text: noteChildren.map(c => c.text || '').join(''),
330
+ children: noteChildren,
331
+ metadata: { noteType: 'footnote', noteId }
332
+ };
333
+ // Notes attach to the preceding text node (matches WordParser's convention);
334
+ // fall back to an empty text node if the reference opens the inline run.
335
+ if (nodes.length > 0) {
336
+ const target = nodes[nodes.length - 1];
337
+ if (!target.notes)
338
+ target.notes = [];
339
+ target.notes.push(noteNode);
103
340
  }
104
341
  else {
105
- const linkNodes = parseInline(altText, currentFormatting);
106
- linkNodes.forEach(n => {
107
- if (n.type === 'text') {
108
- n.metadata = { link: url, linkType: 'external' };
109
- }
110
- });
111
- nodes.push(...linkNodes);
342
+ nodes.push({ type: 'text', text: '', notes: [noteNode] });
112
343
  }
113
344
  }
114
- else if (match[4]) { // Bold
115
- nodes.push(...parseInline(match[4], { ...currentFormatting, bold: true }));
345
+ else if (g.citationKey !== undefined) { // Citation reference
346
+ nodes.push({ type: 'text', text: g.citationKey, metadata: { citationKey: g.citationKey } });
116
347
  }
117
- else if (match[5]) { // Italic
118
- nodes.push(...parseInline(match[5], { ...currentFormatting, italic: true }));
348
+ else if (g.wikiPage !== undefined) { // Wikilink
349
+ const page = g.wikiPage.trim();
350
+ const alias = g.wikiAlias?.trim();
351
+ nodes.push({ type: 'text', text: alias || page, metadata: { link: page, linkType: 'internal', wikilink: true } });
119
352
  }
120
- else if (match[6]) { // Strikethrough
121
- nodes.push(...parseInline(match[6], { ...currentFormatting, strikethrough: true }));
122
- }
123
- else if (match[7]) { // Inline Code
124
- nodes.push({ type: 'text', text: match[7], formatting: { ...currentFormatting, font: 'monospace' } });
353
+ else if (g.refText !== undefined) { // Explicit/collapsed reference link or image: [text][ref] / [text][]
354
+ const isImage = g.refBang === '!';
355
+ const label = g.refText;
356
+ const refId = (g.refId || label).trim().toLowerCase();
357
+ const def = linkDefinitions.get(refId);
358
+ if (def) {
359
+ nodes.push(...buildLinkOrImageNodes(isImage, label, def.url));
360
+ }
361
+ else {
362
+ // Not a known reference - preserve the literal bracketed text unchanged.
363
+ nodes.push(plainText(text.substring(match.index, match.index + match[0].length)));
364
+ }
125
365
  }
126
- else if (match[8]) { // Underline
127
- nodes.push(...parseInline(match[8], { ...currentFormatting, underline: true }));
366
+ else if (g.shortText !== undefined) { // Shortcut reference: [text]
367
+ const isImage = g.shortBang === '!';
368
+ const label = g.shortText;
369
+ const def = linkDefinitions.get(label.trim().toLowerCase());
370
+ if (def) {
371
+ nodes.push(...buildLinkOrImageNodes(isImage, label, def.url));
372
+ }
373
+ else {
374
+ // Not a known reference - ordinary bracketed prose, preserve unchanged.
375
+ nodes.push(plainText(`${g.shortBang}[${label}]`));
376
+ }
128
377
  }
129
- else if (match[9]) { // Subscript
130
- nodes.push(...parseInline(match[9], { ...currentFormatting, subscript: true }));
378
+ else if (g.autolinkUrl !== undefined) { // <url> autolink
379
+ const url = g.autolinkUrl;
380
+ nodes.push({ type: 'text', text: url, formatting: Object.keys(currentFormatting).length > 0 ? { ...currentFormatting } : undefined, metadata: { link: url, linkType: 'external' } });
131
381
  }
132
- else if (match[10]) { // Superscript
133
- nodes.push(...parseInline(match[10], { ...currentFormatting, superscript: true }));
382
+ else if (g.mathInline !== undefined) { // Inline math
383
+ nodes.push({ type: 'code', text: g.mathInline, metadata: { math: 'inline' } });
134
384
  }
135
385
  lastIndex = regex.lastIndex;
136
386
  }
137
387
  if (lastIndex < text.length) {
138
- nodes.push({ type: 'text', text: text.substring(lastIndex), formatting: Object.keys(currentFormatting).length > 0 ? { ...currentFormatting } : undefined });
388
+ nodes.push(plainText(text.substring(lastIndex)));
389
+ }
390
+ return applyAbbreviations(decodeHtmlEntities(nodes));
391
+ };
392
+ const escapeRegExpChars = (s) => s.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
393
+ // A deliberately small, common-entity lookup (not the full HTML5 named-character-
394
+ // reference table) - keeps this a plain object rather than needing a dependency.
395
+ const NAMED_HTML_ENTITIES = {
396
+ amp: '&', lt: '<', gt: '>', quot: '"', apos: '\'', nbsp: ' ',
397
+ copy: '©', reg: '®', mdash: '—', ndash: '–', hellip: '…'
398
+ };
399
+ // Decodes HTML named entities and numeric/hex character references (&#NN;/&#xHH;)
400
+ // in plain text nodes, skipping monospace (inline code) nodes since CommonMark does
401
+ // not decode entities inside code spans. The regex only ever matches syntactically
402
+ // well-formed &name;/&#NN;/&#xHH; tokens to begin with, so ordinary text containing
403
+ // a bare "&" (e.g. "Q&A", "Fish & Chips") never matches at all; an unrecognized-but-
404
+ // well-formed token (e.g. "&foo;") is left untouched on a lookup miss - no risk of
405
+ // double-decoding or corrupting text that merely resembles an entity.
406
+ const decodeHtmlEntities = (nodes) => {
407
+ return nodes.map(node => {
408
+ if (node.type !== 'text' || !node.text || node.formatting?.font === 'monospace')
409
+ return node;
410
+ const text = node.text.replace(/&(#\d+|#[xX][0-9a-fA-F]+|[a-zA-Z][a-zA-Z0-9]*);/g, (full, ref) => {
411
+ if (ref[0] === '#') {
412
+ const codePoint = ref[1].toLowerCase() === 'x' ? parseInt(ref.slice(2), 16) : parseInt(ref.slice(1), 10);
413
+ return (isNaN(codePoint) || codePoint < 0 || codePoint > 0x10FFFF) ? full : String.fromCodePoint(codePoint);
414
+ }
415
+ return NAMED_HTML_ENTITIES[ref] ?? full;
416
+ });
417
+ return text === node.text ? node : { ...node, text };
418
+ });
419
+ };
420
+ // Splits abbreviation occurrences out of plain text nodes so they carry
421
+ // TextMetadata.abbreviationTitle, rendered as <abbr title> in HTML/editor output.
422
+ const applyAbbreviations = (nodes) => {
423
+ if (abbreviationDefinitions.size === 0)
424
+ return nodes;
425
+ const pattern = new RegExp(`\\b(${[...abbreviationDefinitions.keys()].map(escapeRegExpChars).join('|')})\\b`, 'g');
426
+ const result = [];
427
+ for (const node of nodes) {
428
+ if (node.type !== 'text' || !node.text || node.metadata) {
429
+ result.push(node);
430
+ continue;
431
+ }
432
+ let lastIndex = 0;
433
+ let match;
434
+ let matched = false;
435
+ pattern.lastIndex = 0;
436
+ while ((match = pattern.exec(node.text)) !== null) {
437
+ matched = true;
438
+ if (match.index > lastIndex) {
439
+ result.push({ type: 'text', text: node.text.substring(lastIndex, match.index), formatting: node.formatting });
440
+ }
441
+ result.push({
442
+ type: 'text',
443
+ text: match[0],
444
+ formatting: node.formatting,
445
+ metadata: { abbreviationTitle: abbreviationDefinitions.get(match[0]) }
446
+ });
447
+ lastIndex = pattern.lastIndex;
448
+ }
449
+ if (!matched) {
450
+ result.push(node);
451
+ continue;
452
+ }
453
+ if (lastIndex < node.text.length) {
454
+ result.push({ type: 'text', text: node.text.substring(lastIndex), formatting: node.formatting });
455
+ }
139
456
  }
140
- return nodes;
457
+ return result;
458
+ };
459
+ // Splits a paragraph-shaped block's internal lines into inline-parsed content,
460
+ // inserting a real 'break' node for a hard line break (a line ending in 2+ trailing
461
+ // spaces or a trailing backslash) instead of collapsing it to a space. A plain single
462
+ // newline with no such marker is still a soft break and collapses to a space,
463
+ // unchanged from before - CommonMark itself renders a soft break as a space/newline.
464
+ const splitParagraphLines = (block) => {
465
+ const lines = block.split('\n');
466
+ const children = [];
467
+ lines.forEach((line, i) => {
468
+ const hardBreak = /(?: {2,}|\\)$/.test(line);
469
+ children.push(...parseInline(line.replace(/(?: {2,}|\\)$/, '')));
470
+ if (i < lines.length - 1) {
471
+ if (hardBreak) {
472
+ children.push({ type: 'break', metadata: { breakType: 'carriageReturn' } });
473
+ }
474
+ else {
475
+ children.push({ type: 'text', text: ' ' });
476
+ }
477
+ }
478
+ });
479
+ return children;
480
+ };
481
+ // Builds an admonition node from its raw body text, splitting on blank lines into
482
+ // paragraph children. v1 only supports inline content inside admonitions (no nested
483
+ // lists/headings/code) - acceptable per the roadmap's first cut.
484
+ const buildAdmonitionNode = (admonitionType, body, sourceSyntax) => {
485
+ const paragraphs = body.split(/\n\n+/).map(p => p.trim()).filter(Boolean);
486
+ const children = paragraphs.map(p => ({
487
+ type: 'paragraph',
488
+ children: splitParagraphLines(p)
489
+ }));
490
+ return {
491
+ type: 'admonition',
492
+ metadata: { admonitionType, sourceSyntax },
493
+ children
494
+ };
141
495
  };
142
496
  const rawBlocks = textStr.split(/\n\n+/);
143
497
  const blocks = [];
@@ -148,25 +502,39 @@ const parseMarkdown = async (buffer, config) => {
148
502
  // Match headings or lists that might be joined with other text via single newline
149
503
  const lines = rawBlock.split('\n');
150
504
  let currentSubBlock = [];
505
+ // Tracks whether we're currently "inside" a list (a list-item line, or an
506
+ // indented continuation line right after one) so a continuation line doesn't
507
+ // itself get treated as the boundary that splits the list into a new block -
508
+ // see the "Lists" block dispatch below, which merges such a line into the
509
+ // previous item's content instead of dropping it.
510
+ let inList = false;
151
511
  for (const line of lines) {
512
+ (0, errorUtils_js_1.checkAbortSignal)(config.abortSignal);
152
513
  const isHeading = !!line.match(/^(?:<a[^>]*><\/a>)*\s*#{1,6}\s+/);
153
- const isList = !!line.match(/^(\s*)([-*+]|\d+\.)\s+/);
514
+ const isList = !!line.match(/^(\s*)([-*+]|\d+[.)])\s+/);
154
515
  const isHtmlTag = !!line.match(/^<\/?div[^>]*>$/i);
155
- const prevWasList = currentSubBlock.length > 0 && !!currentSubBlock[currentSubBlock.length - 1].match(/^(\s*)([-*+]|\d+\.)\s+/);
516
+ // A non-list, non-blank, indented (>=2 columns or a tab) line encountered
517
+ // while already inside a list is a continuation of the current item, not a
518
+ // new construct. Scoped to a single such line at a time (no nested
519
+ // code/blockquote/sub-list/multi-paragraph items - those require un-splitting
520
+ // already-separated raw blocks, out of scope here).
521
+ const isContinuation = !isList && inList && /^(?: {2,}|\t)/.test(line) && line.trim().length > 0;
522
+ const staysInListMode = isList || isContinuation;
156
523
  // Split if:
157
524
  // 1. Current line is a heading
158
- // 2. Current line is a list item but previous was NOT
159
- // 3. Current line is NOT a list item but previous WAS
160
- // 4. Current line is an HTML tag (div)
161
- if ((isHeading || isHtmlTag || (isList !== prevWasList)) && currentSubBlock.length > 0) {
525
+ // 2. Current line enters or leaves "list mode" relative to the previous line
526
+ // 3. Current line is an HTML tag (div)
527
+ if ((isHeading || isHtmlTag || (staysInListMode !== inList)) && currentSubBlock.length > 0) {
162
528
  blocks.push(currentSubBlock.join('\n'));
163
529
  currentSubBlock = [];
164
530
  }
165
531
  currentSubBlock.push(line);
532
+ inList = staysInListMode;
166
533
  // Headings and HTML tags are single-line blocks for our state machine
167
534
  if (isHeading || isHtmlTag) {
168
535
  blocks.push(currentSubBlock.join('\n'));
169
536
  currentSubBlock = [];
537
+ inList = false;
170
538
  }
171
539
  }
172
540
  if (currentSubBlock.length > 0) {
@@ -176,9 +544,28 @@ const parseMarkdown = async (buffer, config) => {
176
544
  let listIdCounter = 1;
177
545
  let currentAlignment = undefined;
178
546
  for (let block of blocks) {
547
+ // Preserved before the generic trim() below, which strips the first line's
548
+ // leading indentation - the indented-code-block check further down needs every
549
+ // line's original indentation, including the first.
550
+ const untrimmedBlock = block;
179
551
  block = block.trim();
180
552
  if (!block)
181
553
  continue;
554
+ // Standalone anchor-only block: one or more empty `<a name|id="…"></a>` tags on their
555
+ // own line (bookmark targets the MarkdownGenerator emits just before a heading/paragraph).
556
+ // Capture them as a placeholder so the post-loop pass can re-attach them to the following
557
+ // node's anchorIds — otherwise the tag-opening `<` is escaped and they render as visible text.
558
+ if (/^(?:\s*<a\s[^>]*>\s*<\/a>\s*)+$/i.test(block)) {
559
+ const anchorIds = [];
560
+ for (const m of block.matchAll(/<a\s[^>]*\b(?:name|id)="([^"]*)"/gi)) {
561
+ if (m[1])
562
+ anchorIds.push(m[1]);
563
+ }
564
+ if (anchorIds.length > 0) {
565
+ content.push({ type: ANCHOR_PLACEHOLDER, metadata: { anchorIds }, children: [] });
566
+ continue;
567
+ }
568
+ }
182
569
  // Check for alignment wrapper start/end
183
570
  const alignStartMatch = block.match(/^<div\s+(?:style="text-align:\s*(left|center|right|justify);?"|align="(left|center|right|justify)")>$/i);
184
571
  if (alignStartMatch) {
@@ -196,6 +583,32 @@ const parseMarkdown = async (buffer, config) => {
196
583
  alignment = (alignMatch[1] || alignMatch[2]).toLowerCase();
197
584
  block = alignMatch[3];
198
585
  }
586
+ // YouTube embed fallback: MarkdownGenerator's 'embed' case emits a single-line
587
+ // <div data-youtube-video="ID" data-width="…" data-align="…"></div> when fallbackToHtml
588
+ // is on; recognise it here so a saved-then-reopened .md keeps the video.
589
+ const youtubeMatch = block.match(/^<div\s+data-youtube-video="([^"]*)"([^>]*)>\s*<\/div>$/i);
590
+ if (youtubeMatch) {
591
+ const videoId = youtubeMatch[1];
592
+ const attrsStr = youtubeMatch[2];
593
+ const widthMatch = attrsStr.match(/data-width="([^"]*)"/i);
594
+ const youtubeAlignMatch = attrsStr.match(/data-align="([^"]*)"/i);
595
+ const embedAlign = youtubeAlignMatch && ['left', 'center', 'right'].includes(youtubeAlignMatch[1]) ? youtubeAlignMatch[1] : undefined;
596
+ const embedUrl = videoId ? `https://www.youtube.com/watch?v=${videoId}` : undefined;
597
+ content.push({
598
+ type: 'embed',
599
+ // Childless nodes need .text so generic AST consumers (toText, chunking)
600
+ // don't silently drop them.
601
+ text: embedUrl,
602
+ metadata: {
603
+ embedType: 'youtube',
604
+ videoId,
605
+ url: embedUrl,
606
+ width: widthMatch?.[1],
607
+ align: embedAlign
608
+ }
609
+ });
610
+ continue;
611
+ }
199
612
  // Code Block
200
613
  const codeMatch = block.match(/^__CODE_BLOCK_(\d+)__$/);
201
614
  if (codeMatch) {
@@ -207,6 +620,23 @@ const parseMarkdown = async (buffer, config) => {
207
620
  });
208
621
  continue;
209
622
  }
623
+ // GLFM-style fenced-div admonition, extracted to a placeholder above
624
+ const admonitionBlockMatch = block.match(/^__ADMONITION_(\d+)__$/);
625
+ if (admonitionBlockMatch) {
626
+ const data = JSON.parse(admonitionBlocks[parseInt(admonitionBlockMatch[1])]);
627
+ content.push(buildAdmonitionNode(data.admonitionType, data.body, 'gitlab'));
628
+ continue;
629
+ }
630
+ // Block math ($$...$$), extracted to a placeholder above
631
+ const mathBlockMatch = block.match(/^__MATH_BLOCK_(\d+)__$/);
632
+ if (mathBlockMatch) {
633
+ content.push({
634
+ type: 'code',
635
+ text: mathBlocks[parseInt(mathBlockMatch[1])],
636
+ metadata: { math: 'block' }
637
+ });
638
+ continue;
639
+ }
210
640
  // Heading (allowing for leading HTML anchors and trailing {#anchor})
211
641
  const headingMatch = block.match(/^((?:<a[^>]*><\/a>)*)\s*(#{1,6})\s+(.*?)(?:\s+\{#([^}]+)\})?\s*$/s);
212
642
  if (headingMatch) {
@@ -215,7 +645,7 @@ const parseMarkdown = async (buffer, config) => {
215
645
  const explicitAnchor = headingMatch[4];
216
646
  const anchorIds = [];
217
647
  if (leadingAnchorsRaw) {
218
- const idMatches = leadingAnchorsRaw.matchAll(/<a\s+name="([^"]+)"/gi);
648
+ const idMatches = leadingAnchorsRaw.matchAll(/<a\s[^>]*\b(?:name|id)="([^"]+)"/gi);
219
649
  for (const m of idMatches)
220
650
  anchorIds.push(m[1]);
221
651
  }
@@ -234,43 +664,138 @@ const parseMarkdown = async (buffer, config) => {
234
664
  });
235
665
  continue;
236
666
  }
667
+ // Setext heading (Text\n=== or Text\n---): a line of text immediately followed
668
+ // by a lone `=`/`-` underline with no blank line between them. By the time a
669
+ // block reaches this point, the sub-splitter above has already separated out any
670
+ // genuinely blank-line-preceded thematic break into its own isolated block (which
671
+ // has no preceding text line to combine with here), so this only fires for the
672
+ // ambiguous "text directly above a dash/equals-only line" shape setext needs.
673
+ // Scoped to a single line immediately above the underline becoming the heading
674
+ // text; multi-line setext text (CommonMark's "Foo\nbar\n===" merging into one
675
+ // heading) is an explicitly out-of-scope simplification - any earlier lines in
676
+ // the block are pushed as a separate paragraph first.
677
+ const setextMatch = block.match(/^([\s\S]*)\n([=]+|-+)[ \t]*$/);
678
+ if (setextMatch) {
679
+ const lines = setextMatch[1].split('\n');
680
+ const headingLine = lines[lines.length - 1];
681
+ const earlierLines = lines.slice(0, -1).join('\n').trim();
682
+ if (earlierLines) {
683
+ content.push({
684
+ type: 'paragraph',
685
+ metadata: { alignment },
686
+ children: splitParagraphLines(earlierLines)
687
+ });
688
+ }
689
+ const children = parseInline(headingLine);
690
+ content.push({
691
+ type: 'heading',
692
+ text: children.map(c => c.text || '').join(''),
693
+ metadata: { level: setextMatch[2][0] === '=' ? 1 : 2, alignment },
694
+ children
695
+ });
696
+ continue;
697
+ }
237
698
  // Blockquote
238
699
  const quoteMatch = block.match(/^>\s+(.*)$/s);
239
700
  if (quoteMatch) {
701
+ // [ \t]? (not \s+) so a bare ">" paragraph-separator line (used between
702
+ // multi-paragraph admonition bodies) also dequotes to an empty line. Repeat
703
+ // until no line still starts with ">" so arbitrarily-nested blockquotes
704
+ // (`> > quoted`, `> > > quoted`, ...) are fully unwrapped rather than only
705
+ // stripping one level.
706
+ let dequoted = quoteMatch[1];
707
+ while (/^>/m.test(dequoted)) {
708
+ dequoted = dequoted.replace(/^>[ \t]?/gm, '');
709
+ }
710
+ // GitHub-style admonition: `> [!NOTE]` on the first quoted line.
711
+ const admonitionHeaderMatch = dequoted.match(/^\[!(NOTE|TIP|IMPORTANT|WARNING|CAUTION)\]\s*\n?([\s\S]*)$/i);
712
+ if (admonitionHeaderMatch) {
713
+ const admonitionType = admonitionHeaderMatch[1].toLowerCase();
714
+ content.push(buildAdmonitionNode(admonitionType, admonitionHeaderMatch[2], 'github'));
715
+ continue;
716
+ }
240
717
  content.push({
241
718
  type: 'paragraph',
242
719
  metadata: { style: 'Quote' },
243
- children: parseInline(quoteMatch[1].replace(/^>\s+/gm, ''))
720
+ children: parseInline(dequoted)
721
+ });
722
+ continue;
723
+ }
724
+ // Definition list (Markdown Extra / Pandoc / Kramdown): a term line followed by
725
+ // one or more ": definition" lines, e.g.:
726
+ // Term
727
+ // : Definition of the term.
728
+ const definitionListMatch = block.match(/^([^\n:][^\n]*)\n((?::[ \t]+.+(?:\n:[ \t]+.+)*))$/);
729
+ if (definitionListMatch) {
730
+ const term = definitionListMatch[1];
731
+ const definitions = definitionListMatch[2].split('\n').map(line => line.replace(/^:[ \t]+/, ''));
732
+ content.push({
733
+ type: 'definitionList',
734
+ children: [
735
+ { type: 'definitionTerm', children: parseInline(term) },
736
+ ...definitions.map(def => ({ type: 'definitionDescription', children: parseInline(def) }))
737
+ ]
244
738
  });
245
739
  continue;
246
740
  }
247
741
  // Lists
248
- if (block.match(/^(\s*)([-*+]|\d+\.)\s+/)) {
742
+ if (block.match(/^(\s*)([-*+]|\d+[.)])\s+/)) {
249
743
  const lines = block.split('\n');
250
744
  const listId = `md-list-${listIdCounter++}`;
251
- const listCounters = {};
745
+ const listCounters = new Map();
746
+ // Relative indent stack (not a fixed-width divisor) so nesting level is
747
+ // computed from what indentation actually appeared in this block, rather
748
+ // than assuming a specific indent width. This makes the parser agnostic to
749
+ // 2-space (hand-written), 4-space (this generator's own output), or
750
+ // tab-indented (normalized to a 4-column stop) nested lists.
751
+ const indentStack = [];
752
+ // The most recently pushed list-item node, so a following indented
753
+ // continuation line (see the sub-splitter above) can be merged into it
754
+ // instead of being silently dropped.
755
+ let lastListNode;
252
756
  for (const line of lines) {
253
- const match = line.match(/^(\s*)([-*+]|\d+\.)\s+(.*)$/);
757
+ const match = line.match(/^(\s*)([-*+]|\d+[.)])\s+(.*)$/);
254
758
  if (match) {
255
- const indent = match[1].length / 2;
256
- const level = Math.floor(indent);
759
+ const rawIndent = match[1].replace(/\t/g, ' ').length;
760
+ while (indentStack.length > 0 && rawIndent <= indentStack[indentStack.length - 1]) {
761
+ indentStack.pop();
762
+ }
763
+ const level = indentStack.length;
764
+ indentStack.push(rawIndent);
765
+ // Purge any deeper levels' counters now that we're back at this
766
+ // level - otherwise a nested sub-list under a later sibling item
767
+ // would incorrectly continue a previous sibling's child numbering
768
+ // instead of restarting at 0.
769
+ for (const key of [...listCounters.keys()]) {
770
+ if (key > level)
771
+ listCounters.delete(key);
772
+ }
257
773
  const marker = match[2];
258
- const isOrdered = !!marker.match(/\d+\./);
774
+ const isOrdered = !!marker.match(/\d+[.)]/);
259
775
  const listType = isOrdered ? 'ordered' : 'unordered';
260
- if (listCounters[level] === undefined) {
776
+ if (listCounters.get(level) === undefined) {
261
777
  if (isOrdered) {
262
778
  const startNum = parseInt(marker, 10);
263
- listCounters[level] = isNaN(startNum) ? 0 : startNum - 1;
779
+ listCounters.set(level, isNaN(startNum) ? 0 : startNum - 1);
264
780
  }
265
781
  else {
266
- listCounters[level] = 0;
782
+ listCounters.set(level, 0);
267
783
  }
268
784
  }
269
785
  else {
270
- listCounters[level]++;
786
+ listCounters.set(level, listCounters.get(level) + 1);
271
787
  }
272
- const children = parseInline(match[3]);
273
- content.push({
788
+ let itemText = match[3];
789
+ let isTask;
790
+ let checked;
791
+ const taskMatch = itemText.match(/^\[([ xX])\]\s+(.*)$/);
792
+ if (taskMatch) {
793
+ isTask = true;
794
+ checked = taskMatch[1].toLowerCase() === 'x';
795
+ itemText = taskMatch[2];
796
+ }
797
+ const children = parseInline(itemText);
798
+ const listNode = {
274
799
  type: 'list',
275
800
  text: children.map(c => c.text || '').join(''),
276
801
  metadata: {
@@ -278,18 +803,41 @@ const parseMarkdown = async (buffer, config) => {
278
803
  indentation: level,
279
804
  alignment: alignment || 'left',
280
805
  listId,
281
- itemIndex: listCounters[level]
806
+ itemIndex: listCounters.get(level),
807
+ isTask,
808
+ checked
282
809
  },
283
810
  children
284
- });
811
+ };
812
+ content.push(listNode);
813
+ lastListNode = listNode;
814
+ }
815
+ else if (lastListNode && line.trim().length > 0 && /^(?: {2,}|\t)/.test(line)) {
816
+ // Indented continuation line: merge its inline content into the
817
+ // previous item rather than dropping it. Scoped to a single such
818
+ // line (no nested code/blockquote/sub-list/multi-paragraph items).
819
+ const continuationChildren = parseInline(line.trim());
820
+ lastListNode.children = [...(lastListNode.children || []), { type: 'text', text: ' ' }, ...continuationChildren];
821
+ lastListNode.text = (lastListNode.children || []).map(c => c.text || '').join('');
285
822
  }
286
823
  }
287
824
  continue;
288
825
  }
289
826
  // Table (Simple Pipe or HTML)
290
827
  if ((block.includes('|') && block.match(/\n\s*\|?[-:| ]+\|?\s*\n/)) || block.includes('<table')) {
828
+ // Pandoc-style trailing attribute list (`{align=right}`) immediately after the
829
+ // table, or Kramdown's `{: align=right}` on its own following line - both land
830
+ // in this same raw block since there's no blank line separating them.
831
+ let tableAlign;
832
+ const tableAttrLineMatch = block.match(/\n\{:?\s*([^}]*)\}\s*$/);
833
+ if (tableAttrLineMatch) {
834
+ tableAlign = parseAttributeList(tableAttrLineMatch[1]).align;
835
+ block = block.slice(0, tableAttrLineMatch.index);
836
+ }
291
837
  if (block.includes('<table')) {
292
838
  // Basic HTML table recognition (extracting rows/cells)
839
+ const tableTagMatch = block.match(/<table([^>]*)>/i);
840
+ const tableAlignMatch = tableTagMatch?.[1]?.match(/data-align=["']?(left|center|right)["']?/i);
293
841
  const rows = [];
294
842
  const trRegex = /<tr[^>]*>([\s\S]*?)<\/tr>/gi;
295
843
  let trMatch;
@@ -314,8 +862,13 @@ const parseMarkdown = async (buffer, config) => {
314
862
  if (cells.length > 0)
315
863
  rows.push({ type: 'row', children: cells });
316
864
  }
865
+ const resolvedAlign = tableAlign || (tableAlignMatch ? tableAlignMatch[1].toLowerCase() : undefined);
317
866
  if (rows.length > 0) {
318
- content.push({ type: 'table', children: rows });
867
+ content.push({
868
+ type: 'table',
869
+ metadata: resolvedAlign ? { align: resolvedAlign } : undefined,
870
+ children: rows
871
+ });
319
872
  continue;
320
873
  }
321
874
  }
@@ -323,21 +876,50 @@ const parseMarkdown = async (buffer, config) => {
323
876
  const lines = block.trim().split('\n');
324
877
  const rows = [];
325
878
  for (let i = 0; i < lines.length; i++) {
326
- if (lines[i].match(/^\|?[-:| ]*---[-:| ]*\|?$/))
327
- continue; // Separator row (requires at least one triple-hyphen)
879
+ if (lines[i].match(/^\|?\s*:?-+:?\s*(?:\|\s*:?-+:?\s*)*\|?$/))
880
+ continue; // Separator row (per-cell `:?-+:?`, GFM-style; accepts short cells like `|-|-|`)
328
881
  const cellsStr = lines[i].replace(/^\||\|$/g, '').split('|');
329
- const cells = cellsStr.map(c => ({
330
- type: 'cell',
331
- children: parseInline(c.trim(), i === 0 ? { bold: true } : {})
332
- }));
882
+ const cells = cellsStr.map(c => {
883
+ // Recognize the MarkdownGenerator's own cell-alignment fallback,
884
+ // `<div style="text-align: X">…</div>`, and lift it into an aligned
885
+ // paragraph so it round-trips as alignment instead of being escaped to
886
+ // visible text on regeneration. Unwrap wherever it sits (e.g. inside **…**).
887
+ let cellText = c.trim();
888
+ let cellAlign;
889
+ cellText = cellText.replace(/<div\s+style="text-align:\s*(left|center|right|justify);?"\s*>([\s\S]*?)<\/div>/gi, (_m, a, inner) => { cellAlign = a.toLowerCase(); return inner; });
890
+ const inline = parseInline(cellText, i === 0 ? { bold: true } : {});
891
+ if (cellAlign && cellAlign !== 'left') {
892
+ return {
893
+ type: 'cell',
894
+ children: [{ type: 'paragraph', metadata: { alignment: cellAlign }, children: inline }]
895
+ };
896
+ }
897
+ return { type: 'cell', children: inline };
898
+ });
333
899
  rows.push({ type: 'row', children: cells });
334
900
  }
335
- content.push({ type: 'table', children: rows });
901
+ content.push({ type: 'table', metadata: tableAlign ? { align: tableAlign } : undefined, children: rows });
902
+ continue;
903
+ }
904
+ }
905
+ // Indented code block (4-space or tab indent on every non-blank line). Only
906
+ // reaches this point once heading/blockquote/definition-list/list/table have
907
+ // already failed to claim the block; since list continuation lines are now
908
+ // handled inside the "Lists" branch above and the sub-splitter already isolates
909
+ // list/heading content into their own blocks, a block that's uniformly indented
910
+ // here is not a list by construction. A partially-indented block (some lines
911
+ // indented, some not) falls through to Paragraph unchanged.
912
+ {
913
+ const codeLines = untrimmedBlock.split('\n');
914
+ const nonBlankLines = codeLines.filter(l => l.trim().length > 0);
915
+ if (nonBlankLines.length > 0 && nonBlankLines.every(l => /^(?: {4}|\t)/.test(l))) {
916
+ const stripped = codeLines.map(l => l.replace(/^(?: {4}|\t)/, '')).join('\n');
917
+ content.push({ type: 'code', text: stripped });
336
918
  continue;
337
919
  }
338
920
  }
339
921
  // Hr
340
- if (block.match(/^---+|^\*\*\*+|___+$/)) {
922
+ if (block.match(/^---+$|^\*\*\*+$|^___+$/)) {
341
923
  content.push({ type: 'break', metadata: { breakType: 'page' } });
342
924
  continue;
343
925
  }
@@ -345,17 +927,47 @@ const parseMarkdown = async (buffer, config) => {
345
927
  content.push({
346
928
  type: 'paragraph',
347
929
  metadata: { alignment },
348
- children: parseInline(block.replace(/\n/g, ' '))
930
+ children: splitParagraphLines(block)
349
931
  });
350
932
  }
933
+ // Fold standalone anchor placeholders into the following content node's anchorIds so a
934
+ // bookmark target emitted on its own line round-trips as a real anchor. A trailing placeholder
935
+ // with no following node attaches to the previous node instead; if the document is nothing but
936
+ // anchors, they are dropped (there is no node to host them).
937
+ if (content.some(n => n.type === ANCHOR_PLACEHOLDER)) {
938
+ const merged = [];
939
+ let carried = [];
940
+ for (const node of content) {
941
+ if (node.type === ANCHOR_PLACEHOLDER) {
942
+ carried.push(...(node.metadata?.anchorIds || []));
943
+ continue;
944
+ }
945
+ if (carried.length > 0) {
946
+ const meta = node.metadata || (node.metadata = {});
947
+ meta.anchorIds = [...carried, ...(meta.anchorIds || [])];
948
+ carried = [];
949
+ }
950
+ merged.push(node);
951
+ }
952
+ if (carried.length > 0 && merged.length > 0) {
953
+ const last = merged[merged.length - 1].metadata || (merged[merged.length - 1].metadata = {});
954
+ last.anchorIds = [...(last.anchorIds || []), ...carried];
955
+ }
956
+ content.length = 0;
957
+ content.push(...merged);
958
+ }
351
959
  const toTextSync = () => content.map(n => {
352
960
  const getText = (node) => {
353
961
  if (node.type === 'text' || node.type === 'code')
354
962
  return node.text || '';
355
963
  if (node.type === 'break')
356
964
  return '\n';
965
+ // Childless nodes still carry meaningful text - fall back to it instead of
966
+ // silently vanishing from plain-text/RAG-chunk output.
967
+ if (node.type === 'embed')
968
+ return node.metadata?.url || '';
357
969
  if (node.children) {
358
- const isBlock = ['table', 'row', 'list', 'sheet', 'slide'].includes(node.type);
970
+ const isBlock = ['table', 'row', 'list', 'sheet', 'slide', 'admonition', 'definitionList'].includes(node.type);
359
971
  return node.children.map(getText).join(isBlock ? config.newlineDelimiter : '');
360
972
  }
361
973
  return '';