officeparser 7.2.3 → 7.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/README.md +277 -17
  2. package/dist/OfficeConverter.d.ts +1 -1
  3. package/dist/OfficeConverter.js +3 -0
  4. package/dist/OfficeGenerator.js +4 -0
  5. package/dist/OfficeParser.d.ts +2 -0
  6. package/dist/OfficeParser.js +6 -0
  7. package/dist/cli.d.ts +1 -1
  8. package/dist/cli.js +5 -2
  9. package/dist/defaults.js +12 -0
  10. package/dist/generators/BaseGenerator.d.ts +34 -1
  11. package/dist/generators/BaseGenerator.js +98 -0
  12. package/dist/generators/CsvGenerator.d.ts +9 -1
  13. package/dist/generators/CsvGenerator.js +28 -16
  14. package/dist/generators/EpubGenerator.d.ts +43 -0
  15. package/dist/generators/EpubGenerator.js +312 -0
  16. package/dist/generators/HtmlGenerator.d.ts +12 -0
  17. package/dist/generators/HtmlGenerator.js +378 -61
  18. package/dist/generators/MarkdownGenerator.d.ts +28 -5
  19. package/dist/generators/MarkdownGenerator.js +432 -51
  20. package/dist/generators/PdfGenerator.js +32 -0
  21. package/dist/generators/RtfGenerator.js +47 -22
  22. package/dist/generators/TextGenerator.js +98 -11
  23. package/dist/index.d.ts +1 -0
  24. package/dist/index.js +1 -0
  25. package/dist/officeparser.browser.d.ts +427 -20
  26. package/dist/officeparser.browser.iife.js +338 -206
  27. package/dist/officeparser.browser.mjs +346 -214
  28. package/dist/officeparser.browser.slim.d.ts +427 -20
  29. package/dist/officeparser.browser.slim.iife.js +346 -214
  30. package/dist/officeparser.browser.slim.mjs +346 -214
  31. package/dist/parsers/EpubParser.d.ts +8 -0
  32. package/dist/parsers/EpubParser.js +217 -0
  33. package/dist/parsers/ExcelParser.js +2 -0
  34. package/dist/parsers/HtmlParser.js +507 -48
  35. package/dist/parsers/MarkdownParser.js +704 -92
  36. package/dist/parsers/OpenOfficeParser.js +128 -20
  37. package/dist/parsers/PdfParser.js +4 -1
  38. package/dist/parsers/PowerPointParser.js +1 -0
  39. package/dist/parsers/WordParser.js +1 -0
  40. package/dist/sbom.cdx.json +1695 -0
  41. package/dist/types.d.ts +427 -20
  42. package/dist/types.js +8 -0
  43. package/dist/utils/configUtils.js +53 -4
  44. package/dist/utils/errorUtils.js +7 -3
  45. package/dist/utils/sanitize.d.ts +139 -0
  46. package/dist/utils/sanitize.js +318 -0
  47. package/dist/utils/xmlUtils.js +2 -2
  48. package/dist/utils/zipUtils.js +76 -26
  49. package/package.json +16 -12
@@ -1,7 +1,87 @@
1
1
  "use strict";
2
2
  Object.defineProperty(exports, "__esModule", { value: true });
3
3
  exports.MarkdownGenerator = void 0;
4
+ const sanitize_js_1 = require("../utils/sanitize.js");
4
5
  const BaseGenerator_js_1 = require("./BaseGenerator.js");
6
+ const errorUtils_js_1 = require("../utils/errorUtils.js");
7
+ /**
8
+ * Values accepted for an attribute-list `align=`. Matches what `MarkdownParser`'s own
9
+ * `parseAttributeList` allowlists on import (plus `justify`, which HTML sources can supply),
10
+ * so this is lossless for anything the parser produced.
11
+ */
12
+ const MD_ALIGN_VALUES = new Set(['left', 'center', 'right', 'justify']);
13
+ /** A CSS length or percentage - the only shape `width=` legitimately carries. */
14
+ const MD_LENGTH_PATTERN = /^\d+(?:\.\d+)?(?:px|%|em|rem|pt|pc|in|cm|mm|ex|ch|vw|vh)?$/;
15
+ /** Admonition kinds, mirroring the union declared on `AdmonitionMetadata` in types.ts. */
16
+ const MD_ADMONITION_TYPES = new Set(['note', 'tip', 'important', 'warning', 'caution']);
17
+ /**
18
+ * Folds line breaks to spaces.
19
+ *
20
+ * Used on values that sit inside a single-line construct (an abbreviation definition, an
21
+ * admonition's bold title). A raw newline there does not merely look wrong: it terminates the
22
+ * construct and exposes whatever follows as document-level Markdown.
23
+ */
24
+ const foldLines = (value) => String(value ?? '').replace(/[\r\n]+/g, ' ');
25
+ /**
26
+ * Named Markdown dialect presets. `extended` reproduces this library's historical output
27
+ * exactly (every feature on, GitHub-style admonitions) - the backward-compatibility anchor.
28
+ */
29
+ const MARKDOWN_DIALECT_PRESETS = {
30
+ extended: { admonitions: 'github', definitionLists: true, footnotes: true, citations: true, wikilinks: true, math: 'dollar', attributeLists: true, strikethrough: true, bulletListMarker: '-', orderedListMarker: '.', emphasisMarker: 'asterisk', tables: 'native' },
31
+ github: { admonitions: 'github', definitionLists: false, footnotes: true, citations: false, wikilinks: false, math: 'dollar', attributeLists: false, strikethrough: true, bulletListMarker: '-', orderedListMarker: '.', emphasisMarker: 'asterisk', tables: 'native' },
32
+ gitlab: { admonitions: 'gitlab', definitionLists: false, footnotes: true, citations: false, wikilinks: false, math: 'dollar', attributeLists: false, strikethrough: true, bulletListMarker: '-', orderedListMarker: '.', emphasisMarker: 'asterisk', tables: 'native' },
33
+ obsidian: { admonitions: 'github', definitionLists: false, footnotes: true, citations: false, wikilinks: true, math: 'dollar', attributeLists: false, strikethrough: true, bulletListMarker: '-', orderedListMarker: '.', emphasisMarker: 'asterisk', tables: 'native' },
34
+ pandoc: { admonitions: 'pandoc', definitionLists: true, footnotes: true, citations: true, wikilinks: false, math: 'dollar', attributeLists: true, strikethrough: true, bulletListMarker: '-', orderedListMarker: '.', emphasisMarker: 'asterisk', tables: 'native' },
35
+ commonmark: { admonitions: 'none', definitionLists: false, footnotes: false, citations: false, wikilinks: false, math: 'none', attributeLists: false, strikethrough: false, bulletListMarker: '-', orderedListMarker: '.', emphasisMarker: 'asterisk', tables: 'html' },
36
+ };
37
+ /**
38
+ * Normalizes `MdGeneratorConfig.dialect` into a fully-resolved preset. A string names a preset
39
+ * directly; an object's `extends` field (default `'extended'`) names the base preset that any
40
+ * omitted field falls back to - NOT "whatever preset was ambient before", since config merging
41
+ * replaces the whole `dialect` field rather than layering an object on top of a prior string.
42
+ */
43
+ function resolveDialect(dialect) {
44
+ if (dialect === undefined)
45
+ return MARKDOWN_DIALECT_PRESETS.extended;
46
+ if (typeof dialect === 'string')
47
+ return MARKDOWN_DIALECT_PRESETS[dialect] ?? MARKDOWN_DIALECT_PRESETS.extended;
48
+ const base = MARKDOWN_DIALECT_PRESETS[dialect.extends ?? 'extended'] ?? MARKDOWN_DIALECT_PRESETS.extended;
49
+ return {
50
+ admonitions: dialect.admonitions ?? base.admonitions,
51
+ definitionLists: dialect.definitionLists ?? base.definitionLists,
52
+ footnotes: dialect.footnotes ?? base.footnotes,
53
+ citations: dialect.citations ?? base.citations,
54
+ wikilinks: dialect.wikilinks ?? base.wikilinks,
55
+ math: dialect.math ?? base.math,
56
+ attributeLists: dialect.attributeLists ?? base.attributeLists,
57
+ strikethrough: dialect.strikethrough ?? base.strikethrough,
58
+ bulletListMarker: dialect.bulletListMarker ?? base.bulletListMarker,
59
+ orderedListMarker: dialect.orderedListMarker ?? base.orderedListMarker,
60
+ emphasisMarker: dialect.emphasisMarker ?? base.emphasisMarker,
61
+ tables: dialect.tables ?? base.tables,
62
+ };
63
+ }
64
+ /**
65
+ * Normalizes `MdGeneratorConfig.fallbackToHtml` into a fully resolved object, mirroring
66
+ * `HtmlGenerator`'s `resolveStandalone()` pattern: `true`/undefined turns every part on; `false`
67
+ * turns every part off; an object's omitted fields default to on.
68
+ */
69
+ function resolveFallbackToHtml(fallbackToHtml) {
70
+ const uniform = (on) => ({
71
+ textFormatting: on, alignment: on, anchors: on, tables: on, embeds: on, cellLineBreaks: on,
72
+ });
73
+ if (fallbackToHtml === undefined || typeof fallbackToHtml === 'boolean')
74
+ return uniform(fallbackToHtml ?? true);
75
+ const on = uniform(true);
76
+ return {
77
+ textFormatting: fallbackToHtml.textFormatting ?? on.textFormatting,
78
+ alignment: fallbackToHtml.alignment ?? on.alignment,
79
+ anchors: fallbackToHtml.anchors ?? on.anchors,
80
+ tables: fallbackToHtml.tables ?? on.tables,
81
+ embeds: fallbackToHtml.embeds ?? on.embeds,
82
+ cellLineBreaks: fallbackToHtml.cellLineBreaks ?? on.cellLineBreaks,
83
+ };
84
+ }
5
85
  /**
6
86
  * Generates Markdown from an AST.
7
87
  *
@@ -11,11 +91,11 @@ const BaseGenerator_js_1 = require("./BaseGenerator.js");
11
91
  * be used for these features.
12
92
  *
13
93
  * 2. **Fidelity vs. Purity (The `fallbackToHtml` Principle)**:
14
- * - When `fallbackToHtml` is TRUE: The generator prioritizes high-fidelity document
15
- * conversion. It will use HTML tags for features that Markdown cannot natively
16
- * represent (e.g., `<u>` for underline, `<div>` for alignment, `<table>` for
17
- * nested structures or merged cells).
18
- * - When `fallbackToHtml` is FALSE: The generator prioritizes "pure" Markdown.
94
+ * - When a given `fallbackToHtml` part is TRUE: The generator prioritizes high-fidelity
95
+ * document conversion for that part. It will use HTML tags for features that Markdown
96
+ * cannot natively represent (e.g., `<u>` for underline, `<div>` for alignment, `<table>`
97
+ * for nested structures or merged cells).
98
+ * - When FALSE: The generator prioritizes "pure" Markdown for that part.
19
99
  * Unsupported features are either:
20
100
  * - **Skipped**: Non-essential formatting like underline, subscript, superscript,
21
101
  * or text alignment is omitted.
@@ -25,22 +105,82 @@ const BaseGenerator_js_1 = require("./BaseGenerator.js");
25
105
  *
26
106
  * 3. **Consistency**: All similar structural or formatting ideological problems must be
27
107
  * resolved using these same rules to ensure predictable output.
108
+ *
109
+ * 4. **Dialect (`MdGeneratorConfig.dialect`)**: A second, independent axis from `fallbackToHtml` -
110
+ * which *native* Markdown syntax to emit for constructs with more than one real-world
111
+ * convention (admonitions, definition lists, footnotes, citations, wikilinks, math, list/
112
+ * emphasis markers, tables). See `resolveDialect()` and `MARKDOWN_DIALECT_PRESETS` above.
28
113
  */
29
114
  class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
30
115
  isInsideTable = false;
31
116
  hoistedContent = [];
117
+ collectedAbbreviations = new Map();
118
+ resolvedDialect;
119
+ resolvedFallbackToHtml;
32
120
  constructor(ast, config) {
33
121
  super('md', ast, config);
122
+ this.resolvedDialect = resolveDialect(this.config.mdConfig.dialect);
123
+ this.resolvedFallbackToHtml = resolveFallbackToHtml(this.config.mdConfig.fallbackToHtml);
34
124
  }
35
125
  /**
36
126
  * Renders anchor tags if HTML fallback is allowed.
37
127
  */
38
128
  renderAnchors(metadata) {
39
- if (!this.config.mdConfig.fallbackToHtml || this.config.ignoreInternalLinks)
129
+ if (!this.resolvedFallbackToHtml.anchors || this.config.ignoreInternalLinks)
40
130
  return '';
41
131
  const ids = metadata?.anchorIds || [];
42
132
  return ids.map((aid) => `<a id="${this.slugify(aid)}"></a>`).join('');
43
133
  }
134
+ /**
135
+ * Serializes a frontmatter array as a YAML flow sequence (e.g. `[a, b]`), matching
136
+ * MarkdownParser's frontmatter array handling. Plain strings are left bare; anything
137
+ * that would break flow-array syntax (or isn't a string) falls back to JSON encoding.
138
+ */
139
+ serializeFrontmatterArray(arr) {
140
+ const items = arr.map(item => (typeof item === 'string' && item.trim() === item && !/[,[\]]/.test(item))
141
+ ? item
142
+ : JSON.stringify(item));
143
+ return `[${items.join(', ')}]`;
144
+ }
145
+ /**
146
+ * Renders a Pandoc-style attribute list (e.g. `{width=50% align=left}`) from
147
+ * ImageMetadata/TableMetadata's width/align fields - the canonical form is always
148
+ * `key=value`, matching MarkdownParser's own vocabulary (MARKDOWN_DIALECT.md §15).
149
+ */
150
+ renderAttributeList(meta) {
151
+ if (!this.resolvedDialect.attributeLists)
152
+ return '';
153
+ if (!meta?.width && !meta?.align)
154
+ return '';
155
+ const parts = [];
156
+ // Allowlist, not escape. These land in `metadata.width`/`align` on reparse, which the
157
+ // parser does NOT entity-decode, so encoding here would not round-trip - and stripping
158
+ // alone is not enough: the previous `[{}\s]+` guard removed whitespace, which stops
159
+ // `<img src=x onerror=…>` but not the slash-separated `<img/src=x/onerror=…>`.
160
+ // Both values have a small, fully-known shape, so matching that shape is both safer and
161
+ // lossless for anything a parser can produce.
162
+ //
163
+ // (`isValidContainerWidth` in utils/configUtils.ts is a near-identical regex, but it is a
164
+ // config validator that also accepts 'auto' and numbers; importing configUtils here for
165
+ // one pattern would be a worse coupling than this local constant.)
166
+ if (meta.width && MD_LENGTH_PATTERN.test(String(meta.width).trim())) {
167
+ parts.push(`width=${String(meta.width).trim()}`);
168
+ }
169
+ if (meta.align && MD_ALIGN_VALUES.has(String(meta.align).trim().toLowerCase())) {
170
+ parts.push(`align=${String(meta.align).trim().toLowerCase()}`);
171
+ }
172
+ if (parts.length === 0)
173
+ return '';
174
+ return `{${parts.join(' ')}}`;
175
+ }
176
+ /** Converts a document-supplied date to an ISO string, or '' if invalid
177
+ * (a malformed date would otherwise throw a RangeError and abort generation). */
178
+ toIsoDate(value) {
179
+ if (value === undefined || value === null || value === '')
180
+ return '';
181
+ const d = new Date(value);
182
+ return isNaN(d.getTime()) ? '' : d.toISOString();
183
+ }
44
184
  /**
45
185
  * Generates Markdown string from the provided AST.
46
186
  *
@@ -49,21 +189,34 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
49
189
  async generate() {
50
190
  let output = '';
51
191
  // Add Metadata (YAML Front Matter)
52
- if (this.ast.metadata) {
192
+ const meta = this.effectiveMetadata;
193
+ if (meta) {
53
194
  output += '---\n';
54
- if (this.ast.metadata.title)
55
- output += `title: "${this.ast.metadata.title}"\n`;
56
- if (this.ast.metadata.author)
57
- output += `author: "${this.ast.metadata.author}"\n`;
58
- if (this.ast.metadata.created)
59
- output += `created: ${new Date(this.ast.metadata.created).toISOString()}\n`;
60
- if (this.ast.metadata.modified)
61
- output += `modified: ${new Date(this.ast.metadata.modified).toISOString()}\n`;
62
- if (this.ast.metadata.description)
63
- output += `description: "${this.ast.metadata.description}"\n`;
64
- if (this.ast.metadata.customProperties) {
65
- for (const [key, val] of Object.entries(this.ast.metadata.customProperties)) {
66
- output += `${key}: ${JSON.stringify(val)}\n`;
195
+ // JSON-encode scalar values so a title/author/description containing a
196
+ // quote or newline can't break out of the YAML string and inject
197
+ // arbitrary front-matter keys. (JSON.stringify of a benign value yields
198
+ // the same `"..."` form as before, so normal output is unchanged.)
199
+ if (meta.title)
200
+ output += `title: ${JSON.stringify(meta.title)}\n`;
201
+ if (meta.author)
202
+ output += `author: ${JSON.stringify(meta.author)}\n`;
203
+ const createdIso = this.toIsoDate(meta.created);
204
+ if (createdIso)
205
+ output += `created: ${createdIso}\n`;
206
+ const modifiedIso = this.toIsoDate(meta.modified);
207
+ if (modifiedIso)
208
+ output += `modified: ${modifiedIso}\n`;
209
+ if (meta.description)
210
+ output += `description: ${JSON.stringify(meta.description)}\n`;
211
+ if (meta.subject)
212
+ output += `subject: ${JSON.stringify(meta.subject)}\n`;
213
+ if (meta.keywords)
214
+ output += `keywords: ${JSON.stringify(meta.keywords)}\n`;
215
+ if (meta.customProperties) {
216
+ for (const [key, val] of Object.entries(meta.customProperties)) {
217
+ // Strip newlines/colons from the key so it can't inject a new mapping.
218
+ const safeKey = String(key).replace(/[\r\n:]+/g, ' ').trim();
219
+ output += `${safeKey}: ${Array.isArray(val) ? this.serializeFrontmatterArray(val) : JSON.stringify(val)}\n`;
67
220
  }
68
221
  }
69
222
  output += '---\n\n';
@@ -87,16 +240,19 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
87
240
  }
88
241
  switch (node.type) {
89
242
  case 'text': {
90
- let text = node.text || '';
243
+ // Entity-encode angle brackets so document text can't inject a raw
244
+ // HTML tag (e.g. <script>) when the Markdown is rendered to HTML.
245
+ let text = (0, sanitize_js_1.markdownEscapeText)(node.text || '');
91
246
  if (this.config.includeFormatting && node.formatting) {
247
+ const emphasisAsterisk = this.resolvedDialect.emphasisMarker === 'asterisk';
92
248
  if (node.formatting.bold)
93
- text = `**${text}**`;
249
+ text = emphasisAsterisk ? `**${text}**` : `__${text}__`;
94
250
  if (node.formatting.italic)
95
- text = `*${text}*`;
96
- if (node.formatting.strikethrough)
251
+ text = emphasisAsterisk ? `*${text}*` : `_${text}_`;
252
+ if (node.formatting.strikethrough && this.resolvedDialect.strikethrough)
97
253
  text = `~~${text}~~`;
98
254
  // Use HTML tags for formatting not natively supported by standard Markdown
99
- if (this.config.mdConfig.fallbackToHtml) {
255
+ if (this.resolvedFallbackToHtml.textFormatting) {
100
256
  if (node.formatting.underline)
101
257
  text = `<u>${text}</u>`;
102
258
  if (node.formatting.subscript)
@@ -106,18 +262,52 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
106
262
  }
107
263
  }
108
264
  const meta = node.metadata;
109
- if (meta?.link) {
265
+ if (meta?.wikilink && this.resolvedDialect.wikilinks) {
266
+ // Obsidian syntax: bare page name, or page|alias when the display
267
+ // text differs from the page name. Strip the `[]|`/newline chars
268
+ // that would break out of the `[[...]]` wrapper.
269
+ // The alias must be built from the ESCAPED text, not from raw node.text.
270
+ // Rebuilding from the raw value here discarded the markdownEscapeText()
271
+ // applied above, so a wikilink was the one place document text reached
272
+ // the output unescaped. Escaping is lossless for the alias specifically,
273
+ // because it lands back in a text node, which the parser entity-decodes.
274
+ const alias = (0, sanitize_js_1.markdownEscapeText)(node.text || '').replace(/[[\]|\r\n]+/g, '');
275
+ // `page` lands in metadata.link, which is NOT entity-decoded on reparse,
276
+ // so it gets `<` dropped rather than encoded - a page name is an
277
+ // identifier, and `<` carries no meaning in one.
278
+ const page = (meta.link || '').replace(/[[\]|<\r\n]+/g, '');
279
+ text = (node.text && node.text !== (meta.link || '')) ? `[[${page}|${alias}]]` : `[[${page}]]`;
280
+ }
281
+ else if (meta?.link) {
110
282
  const isInternal = meta.linkType !== 'external';
111
283
  if (!this.config.ignoreInternalLinks || !isInternal) {
112
284
  let link = meta.link;
113
285
  // Slugify internal link targets to match heading IDs if generating IDs
114
- if (isInternal && link.startsWith('#') && (this.config.generateIds || this.config.mdConfig.fallbackToHtml)) {
286
+ if (isInternal && link.startsWith('#') && (this.config.generateIds || this.resolvedFallbackToHtml.anchors)) {
115
287
  const target = link.substring(1);
116
288
  link = '#' + this.slugify(target);
117
289
  }
118
- text = `[${text}](${link})`;
290
+ // Reject javascript:/data: schemes and encode `()`/whitespace so the
291
+ // URL can't break out of `](...)` or inject a script link.
292
+ text = `[${text}](${(0, sanitize_js_1.sanitizeMarkdownUrl)(link)})`;
119
293
  }
120
294
  }
295
+ if (meta?.abbreviationTitle) {
296
+ // Markdown Extra's abbreviation syntax has no inline marker - the bare
297
+ // word round-trips as-is, with its expansion collected at the document
298
+ // end via `*[abbr]: title`.
299
+ this.collectedAbbreviations.set(node.text || '', meta.abbreviationTitle);
300
+ }
301
+ if (meta?.citationKey) {
302
+ // Allowlist to exactly the character class MarkdownParser's own citation
303
+ // recognizer accepts, so this is provably lossless for anything it
304
+ // produced - while fully neutralizing a key arriving from HtmlParser's
305
+ // `data-citation-key`, which accepts any string. Like the wikilink above,
306
+ // this branch also replaces `text` wholesale, so a strip that left `<`
307
+ // behind discarded the escaping applied earlier.
308
+ const key = String(meta.citationKey).replace(/[^a-zA-Z0-9_:.-]/g, '');
309
+ text = this.resolvedDialect.citations ? `[@${key}]` : `[${key}]`;
310
+ }
121
311
  return text;
122
312
  }
123
313
  case 'heading': {
@@ -136,14 +326,14 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
136
326
  else if (this.config.generateIds) {
137
327
  id = ` {#${this.slugify(this.getNodeText(node))}}`;
138
328
  }
139
- const anchors = this.config.mdConfig.fallbackToHtml
140
- ? remainingAnchors.map(aid => `<a name="${aid}"></a>`).join('')
329
+ const anchors = this.resolvedFallbackToHtml.anchors
330
+ ? remainingAnchors.map(aid => `<a name="${this.slugify(aid)}"></a>`).join('')
141
331
  : '';
142
332
  let content = `${prefix}${childrenOutput}${id}`;
143
333
  // Alignment fallback via HTML div/p
144
- if (this.config.mdConfig.fallbackToHtml && meta?.alignment && meta.alignment !== 'left') {
334
+ if (this.resolvedFallbackToHtml.alignment && meta?.alignment && meta.alignment !== 'left') {
145
335
  // Use extra newlines to ensure Markdown inside the div is parsed
146
- content = `<div style="text-align: ${meta.alignment}">\n\n${content}\n\n</div>`;
336
+ content = `<div style="text-align: ${(0, sanitize_js_1.sanitizeCssValue)(meta.alignment)}">\n\n${content}\n\n</div>`;
147
337
  }
148
338
  return `${anchors}${anchors ? '\n' : ''}${content}\n\n`;
149
339
  }
@@ -152,8 +342,8 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
152
342
  const anchors = this.renderAnchors(meta);
153
343
  let content = childrenOutput;
154
344
  // Alignment fallback via HTML div/p
155
- if (this.config.mdConfig.fallbackToHtml && meta?.alignment && meta.alignment !== 'left') {
156
- content = `<div style="text-align: ${meta.alignment}">${content}</div>`;
345
+ if (this.resolvedFallbackToHtml.alignment && meta?.alignment && meta.alignment !== 'left') {
346
+ content = `<div style="text-align: ${(0, sanitize_js_1.sanitizeCssValue)(meta.alignment)}">${content}</div>`;
157
347
  }
158
348
  return childrenOutput ? `${anchors}${content}\n\n` : '';
159
349
  }
@@ -161,7 +351,10 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
161
351
  const meta = node.metadata;
162
352
  const indentSpaces = ' '.repeat(4);
163
353
  const indent = indentSpaces.repeat(meta?.indentation || 0);
164
- const marker = meta?.listType === 'ordered' ? `${(meta.itemIndex ?? 0) + 1}. ` : '- ';
354
+ const bullet = `${this.resolvedDialect.bulletListMarker} `;
355
+ const marker = meta?.isTask
356
+ ? (meta.checked ? `${bullet}[x] ` : `${bullet}[ ] `)
357
+ : (meta?.listType === 'ordered' ? `${(meta.itemIndex ?? 0) + 1}${this.resolvedDialect.orderedListMarker} ` : bullet);
165
358
  const anchors = this.renderAnchors(meta);
166
359
  return `${indent}${marker}${anchors}${childrenOutput}\n`;
167
360
  }
@@ -179,11 +372,25 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
179
372
  }
180
373
  }
181
374
  const anchors = this.renderAnchors(meta);
182
- return `${anchors}${anchors ? '\n' : ''}![${alt}](${src})`;
375
+ // Strip `[]` from alt (would close the `![...]`) and neutralize the URL scheme.
376
+ const safeAlt = (0, sanitize_js_1.markdownEscapeText)(alt).replace(/[[\]]/g, '');
377
+ const safeSrc = (0, sanitize_js_1.sanitizeMarkdownUrl)(src, { allowDataImage: true });
378
+ return `${anchors}${anchors ? '\n' : ''}![${safeAlt}](${safeSrc})${this.renderAttributeList(meta)}`;
183
379
  }
184
380
  case 'table': {
185
381
  const anchors = this.renderAnchors(node.metadata);
186
382
  const tableOutput = await this.renderMarkdownTable(node, processor);
383
+ // The HTML-fallback path (merged cells/nested tables, or a dialect that forces
384
+ // HTML tables outright) already carries data-align on the <table> tag directly -
385
+ // only the plain pipe-table form needs the attribute-list syntax for alignment.
386
+ const usedHtmlFallback = this.resolvedDialect.tables === 'html' ||
387
+ (this.resolvedFallbackToHtml.tables && (this.hasNestedTable(node) || this.hasColspanOrRowspan(node)));
388
+ const attrList = usedHtmlFallback ? '' : this.renderAttributeList(node.metadata);
389
+ if (attrList) {
390
+ // Must glue directly below the last row with no blank line, or
391
+ // MarkdownParser's block splitter won't see it as part of the same block.
392
+ return `${anchors}${anchors ? '\n' : ''}${tableOutput.replace(/\n+$/, '\n')}${attrList}\n`;
393
+ }
187
394
  return `${anchors}${anchors ? '\n' : ''}${tableOutput}`;
188
395
  }
189
396
  case 'row':
@@ -192,17 +399,60 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
192
399
  return childrenOutput;
193
400
  }
194
401
  case 'break': {
402
+ // A hard line break (CommonMark: two trailing spaces before the
403
+ // newline) round-trips back to a distinct 'break' node on reparse;
404
+ // every other breakType (including 'page', used for a thematic-break
405
+ // HR) keeps emitting a bare newline, unchanged.
406
+ const meta = node.metadata;
407
+ if (meta?.breakType === 'carriageReturn')
408
+ return ' \n';
195
409
  return '\n';
196
410
  }
197
411
  case 'code': {
198
412
  const meta = node.metadata;
199
- const lang = meta?.language || '';
200
- // Block code if it contains newlines, else inline
201
- if (node.text && node.text.includes('\n')) {
202
- return `\n\`\`\`${lang}\n${node.text}\n\`\`\`\n\n`;
413
+ // Math content reached the output completely raw, which mattered most under
414
+ // `math: 'none'` (the commonmark preset), where there is no `$` wrapper at
415
+ // all and the text lands directly in the document body.
416
+ //
417
+ // Encode rather than drop: `$a < b$` is ordinary LaTeX, and dropping `<`
418
+ // would silently corrupt real formulae. markdownEscapeText only touches `<`
419
+ // followed by a letter/`/`/`!`/`?`, which is not idiomatic math, and it is
420
+ // idempotent - so output is stable across repeated round-trips even though
421
+ // the first cycle shifts an anomalous `<img` to `&lt;img`. (Fully lossless
422
+ // would mean teaching MarkdownParser.decodeHtmlEntities to cover math `code`
423
+ // nodes; that is a parser behaviour change with its own baseline
424
+ // consequences and must not gate a security fix.)
425
+ if (meta?.math === 'block') {
426
+ // A content line of exactly `$$` would close the block early.
427
+ const mathBlock = (0, sanitize_js_1.markdownEscapeText)(node.text || '')
428
+ .split('\n').map(l => (l.trim() === '$$' ? ` ${l}` : l)).join('\n');
429
+ return this.resolvedDialect.math === 'dollar' ? `\n$$\n${mathBlock}\n$$\n\n` : `\n${mathBlock}\n\n`;
430
+ }
431
+ if (meta?.math === 'inline') {
432
+ // Dropping `$` and newlines is lossless here: the parser's own inline-math
433
+ // recognizer is `\$(?!\s)([^$\n]+?)(?<!\s)\$`, which can never capture either.
434
+ const mathInline = (0, sanitize_js_1.markdownEscapeText)(node.text || '').replace(/[$\r\n]+/g, '');
435
+ return this.resolvedDialect.math === 'dollar' ? `$${mathInline}$` : mathInline;
436
+ }
437
+ const lang = (meta?.language || '').replace(/[\r\n`]+/g, '');
438
+ // Block code if it contains a line break, else inline. Testing only for `\n`
439
+ // routed a CR-only string to the inline branch, where a renderer that
440
+ // normalizes `\r` to a line ending sees a blank line, the span dies, and the
441
+ // remainder is exposed as raw Markdown. The fence sizing below is correct and
442
+ // needs no change; code content itself is not an HTML context.
443
+ if (node.text && /[\r\n]/.test(node.text)) {
444
+ // Fence with one more backtick than the longest run inside the content
445
+ // so an embedded ``` can't close the block early and inject markup.
446
+ const longestRun = Math.max(0, ...(node.text.match(/`+/g) || []).map(s => s.length));
447
+ const fence = '`'.repeat(Math.max(3, longestRun + 1));
448
+ return `\n${fence}${lang}\n${node.text}\n${fence}\n\n`;
203
449
  }
204
450
  else {
205
- return `\`${node.text || ''}\` `;
451
+ const t = node.text || '';
452
+ const longestRun = Math.max(0, ...(t.match(/`+/g) || []).map(s => s.length));
453
+ const fence = '`'.repeat(Math.max(1, longestRun + 1));
454
+ const pad = (t.startsWith('`') || t.endsWith('`')) ? ' ' : '';
455
+ return `${fence}${pad}${t}${pad}${fence} `;
206
456
  }
207
457
  }
208
458
  case 'sheet': {
@@ -220,9 +470,80 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
220
470
  }
221
471
  case 'note': {
222
472
  const meta = node.metadata;
223
- const typeLabel = meta?.noteType === 'footnote' ? 'Footnote' : (meta?.noteType === 'endnote' ? 'Endnote' : 'Note');
224
- return `> **${typeLabel}:** ${childrenOutput.trim()}\n\n`;
473
+ if (meta?.noteType === 'footnote' || meta?.noteType === 'endnote') {
474
+ if (!this.resolvedDialect.footnotes) {
475
+ // Dialect has no footnote syntax - the caller inlines this bare body
476
+ // as a parenthetical at the reference point instead of collecting it
477
+ // into an end-of-document "### Notes" section under a [^id] marker.
478
+ return childrenOutput.trim();
479
+ }
480
+ return `[^${this.getFootnoteKey(node)}]: ${childrenOutput.trim()}\n\n`;
481
+ }
482
+ return `> **Note:** ${childrenOutput.trim()}\n\n`;
225
483
  }
484
+ case 'embed': {
485
+ // Markdown has no native embed syntax. When fallbackToHtml.embeds is on (our
486
+ // save default), emit the exact single-line div MarkdownParser recognises on
487
+ // reimport; otherwise degrade to a plain link.
488
+ const meta = node.metadata;
489
+ const id = meta?.videoId || '';
490
+ if (this.resolvedFallbackToHtml.embeds) {
491
+ const width = meta?.width ? ` data-width="${(0, sanitize_js_1.escapeHtml)(meta.width)}"` : '';
492
+ const align = meta?.align ? ` data-align="${(0, sanitize_js_1.escapeHtml)(meta.align)}"` : '';
493
+ return `\n<div data-youtube-video="${(0, sanitize_js_1.escapeHtml)(id)}"${width}${align}></div>\n\n`;
494
+ }
495
+ const url = meta?.url || (id ? `https://youtu.be/${id}` : '');
496
+ return url ? `[YouTube](${(0, sanitize_js_1.sanitizeMarkdownUrl)(url)})\n\n` : '';
497
+ }
498
+ case 'admonition': {
499
+ const meta = node.metadata;
500
+ // `admonitionType` is a closed union in types.ts and both parsers already
501
+ // allowlist on import, so enforcing it here is a no-op for any conforming
502
+ // AST - it closes the gap for a programmatically-built one, where the type is
503
+ // interpolated straight into `:::TYPE` / `::: {.TYPE}` / `[!TYPE]`.
504
+ const rawType = String(meta?.admonitionType || 'note').toLowerCase();
505
+ const type = MD_ADMONITION_TYPES.has(rawType) ? rawType : 'note';
506
+ const label = type.toUpperCase();
507
+ // A newline in the title would close the `**...**` and, in the fenced-div
508
+ // branches, could emit a stray `:::` line. `title` is never parser-set, so
509
+ // there is no round-trip to preserve and escaping is free.
510
+ const title = meta?.title ? (0, sanitize_js_1.markdownEscapeText)(foldLines(meta.title)) : '';
511
+ const body = childrenOutput.trim();
512
+ switch (this.resolvedDialect.admonitions) {
513
+ case 'gitlab':
514
+ // GLFM fenced-div: no dedicated title syntax, so a custom title (if
515
+ // any) is folded into the body as a bold first line.
516
+ return `:::${type}\n${title ? `**${title}**\n\n` : ''}${body}\n:::\n\n`;
517
+ case 'pandoc':
518
+ // Pandoc's own fenced-div-with-class syntax; same title handling as gitlab.
519
+ return `::: {.${type}}\n${title ? `**${title}**\n\n` : ''}${body}\n:::\n\n`;
520
+ case 'none': {
521
+ // Degrade to a plain bold-labeled blockquote, no special marker.
522
+ const quotedLines = body.split('\n').map(l => l.length > 0 ? `> ${l}` : '>').join('\n');
523
+ const heading = title || label.charAt(0) + label.slice(1).toLowerCase();
524
+ return `> **${heading}:**\n${quotedLines}\n\n`;
525
+ }
526
+ case 'github':
527
+ default: {
528
+ // Canonical GitHub blockquote form. No dedicated title syntax either
529
+ // (matches this library's historical output).
530
+ const quotedLines = body.split('\n').map(l => l.length > 0 ? `> ${l}` : '>').join('\n');
531
+ return `> [!${label}]\n${quotedLines}\n\n`;
532
+ }
533
+ }
534
+ }
535
+ case 'definitionList':
536
+ if (!this.resolvedDialect.definitionLists)
537
+ return `${childrenOutput}\n`;
538
+ return `${childrenOutput}\n`;
539
+ case 'definitionTerm':
540
+ if (!this.resolvedDialect.definitionLists)
541
+ return `**${childrenOutput}**\n\n`;
542
+ return `${childrenOutput}\n`;
543
+ case 'definitionDescription':
544
+ if (!this.resolvedDialect.definitionLists)
545
+ return `${childrenOutput}\n\n`;
546
+ return `: ${childrenOutput}\n`;
226
547
  default:
227
548
  return childrenOutput;
228
549
  }
@@ -253,8 +574,24 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
253
574
  }
254
575
  output += notesMd;
255
576
  }
577
+ if (this.collectedAbbreviations.size > 0) {
578
+ output += '\n\n';
579
+ for (const [abbr, title] of this.collectedAbbreviations) {
580
+ output += `*[${(0, sanitize_js_1.markdownEscapeText)(String(abbr).replace(/[[\]\r\n]+/g, ''))}]: ${(0, sanitize_js_1.markdownEscapeText)(foldLines(title))}\n`;
581
+ }
582
+ }
583
+ // Only a run of literal "\n" at either end is ever a generator artifact here: block
584
+ // separators, the notes/abbreviations sections, the unconditional '\n\n' before
585
+ // hoistedContent (added even when hoistedContent is empty), and renderMarkdownTable's
586
+ // HTML-fallback branches, which unconditionally wrap in a leading+trailing '\n' as
587
+ // separators from whatever precedes/follows (in practice this rarely surfaces at the very
588
+ // start of `output` today since frontmatter's own "---" almost always precedes real
589
+ // content first - see the type doc on `ast.metadata` - but the strip is correct regardless
590
+ // of what precedes it). Nothing else at either end is a generator artifact: not leading
591
+ // whitespace, and not any other kind of trailing whitespace, both of which would be real
592
+ // document content. See the identical reasoning in TextGenerator.generate().
256
593
  return {
257
- value: (output + '\n\n' + this.hoistedContent.join('\n\n')).trim(),
594
+ value: (output + '\n\n' + this.hoistedContent.join('\n\n')).replace(/^\n+|\n+$/g, ''),
258
595
  messages: this.messages
259
596
  };
260
597
  }
@@ -263,6 +600,10 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
263
600
  * Overridden to provide AST optimization (merging adjacent text nodes).
264
601
  */
265
602
  async processNodeRecursive(node, processor) {
603
+ // Mirrors the check in BaseGenerator.processNodeRecursive. This override replaces that
604
+ // method entirely, so without repeating the check here the signal would be silently
605
+ // inert for this generator - which is exactly how it was missed.
606
+ (0, errorUtils_js_1.checkAbortSignal)(this.config.abortSignal);
266
607
  // Allow user to completely override rendering or skip via onNode
267
608
  const override = await this.handleOnNode(node);
268
609
  if (override === false) {
@@ -279,9 +620,16 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
279
620
  childrenOutput += await this.processNodeRecursive(child, processor);
280
621
  }
281
622
  }
623
+ // When the dialect has no footnote syntax, a footnote/endnote is inlined right at its
624
+ // reference point instead (see below) - so it must not also be collected into the
625
+ // end-of-document "### Notes" section, or its content would be duplicated.
626
+ const isInlinedFootnote = (note) => {
627
+ const meta = note.metadata;
628
+ return (meta?.noteType === 'footnote' || meta?.noteType === 'endnote') && !this.resolvedDialect.footnotes;
629
+ };
282
630
  if (node.notes && node.notes.length > 0) {
283
631
  if (node.type !== 'slide') {
284
- this.collectedNotes.push(...node.notes);
632
+ this.collectedNotes.push(...node.notes.filter(note => !isInlinedFootnote(note)));
285
633
  }
286
634
  }
287
635
  let result = await processor(node, childrenOutput);
@@ -290,6 +638,27 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
290
638
  result += await this.processNodeRecursive(note, processor);
291
639
  }
292
640
  }
641
+ else if (node.notes && node.notes.length > 0) {
642
+ for (const note of node.notes) {
643
+ const meta = note.metadata;
644
+ if (meta?.noteType !== 'footnote' && meta?.noteType !== 'endnote')
645
+ continue;
646
+ if (isInlinedFootnote(note)) {
647
+ // Markdown-specific degrade (not RTF/plain-text's "drop the marker, just
648
+ // append at the end" convention): inline the note's rendered body as a
649
+ // parenthetical right where it's referenced, since Markdown readers benefit
650
+ // from an inline association those simpler formats don't need in the same way.
651
+ const body = await this.processNodeRecursive(note, processor);
652
+ result += ` (Note: ${body})`;
653
+ }
654
+ else {
655
+ // Emit the [^id] reference marker at the point of reference. Without this,
656
+ // a footnote/endnote would only ever show up in the collected ### Notes
657
+ // section at the end, with no indication of where it was originally cited.
658
+ result += `[^${this.getFootnoteKey(note)}]`;
659
+ }
660
+ }
661
+ }
293
662
  return result;
294
663
  }
295
664
  /**
@@ -337,13 +706,19 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
337
706
  async renderMarkdownTable(node, processor) {
338
707
  if (!node.children || node.children.length === 0)
339
708
  return '';
709
+ // A dialect that has no native table syntax at all (e.g. strict CommonMark) always
710
+ // renders as HTML, regardless of complexity - this is a separate axis from the
711
+ // nested/merged-cell HTML fallback below, which only applies to otherwise-native tables.
712
+ if (this.resolvedDialect.tables === 'html') {
713
+ return '\n' + await this.renderTableAsHtml(node) + '\n';
714
+ }
340
715
  // If table is complex, nested, or uses merges, fallback to HTML for high fidelity if allowed
341
716
  const isComplex = this.hasNestedTable(node) || this.hasColspanOrRowspan(node);
342
- if (this.config.mdConfig.fallbackToHtml && isComplex) {
717
+ if (this.resolvedFallbackToHtml.tables && isComplex) {
343
718
  return '\n' + await this.renderTableAsHtml(node) + '\n';
344
719
  }
345
720
  // Handle nested tables in pure Markdown by hoisting them out
346
- if (this.isInsideTable && !this.config.mdConfig.fallbackToHtml) {
721
+ if (this.isInsideTable && !this.resolvedFallbackToHtml.tables) {
347
722
  const wasInside = this.isInsideTable;
348
723
  this.isInsideTable = false; // Reset to allow rendering the hoisted table correctly
349
724
  const hoistedId = this.hoistedContent.length + 1;
@@ -384,7 +759,7 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
384
759
  // Process cell content
385
760
  let cellContent = await this.processNodeRecursive(cellNode, processor);
386
761
  // Use <br> fallback only if allowed, otherwise space
387
- const br = this.config.mdConfig.fallbackToHtml ? '<br>' : ' ';
762
+ const br = this.resolvedFallbackToHtml.cellLineBreaks ? '<br>' : ' ';
388
763
  cellContent = cellContent.trim().replace(/\n+/g, br).replace(/\|/g, '\\|');
389
764
  rowCells.push(cellContent);
390
765
  // Handle colspan by adding empty cells
@@ -460,7 +835,11 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
460
835
  rows += await this.renderTableAsHtml(row, await this.handleOnNode(row));
461
836
  }
462
837
  }
463
- return `<table>\n${rows}</table>\n`;
838
+ // Carry table-layout alignment through the HTML fallback so it isn't lost
839
+ // just because the table also needed HTML for merged cells.
840
+ const tableMeta = node.metadata;
841
+ const alignAttr = tableMeta?.align ? ` data-align="${(0, sanitize_js_1.escapeHtml)(tableMeta.align)}"` : '';
842
+ return `<table${alignAttr}>\n${rows}</table>\n`;
464
843
  }
465
844
  else if (node.type === 'row') {
466
845
  let cells = '';
@@ -482,7 +861,9 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
482
861
  content += await this.processNodeRecursive(child, async (n, co) => {
483
862
  switch (n.type) {
484
863
  case 'text': {
485
- let text = n.text || '';
864
+ // Inside HTML table cells, entity-encode angle brackets so cell
865
+ // text can't inject a raw tag (e.g. </td><script>).
866
+ let text = (0, sanitize_js_1.markdownEscapeText)(n.text || '');
486
867
  if (n.formatting?.bold)
487
868
  text = `<b>${text}</b>`;
488
869
  if (n.formatting?.italic)
@@ -497,7 +878,7 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
497
878
  }
498
879
  case 'paragraph': return `<p>${co}</p>`;
499
880
  case 'heading': {
500
- const level = n.metadata?.level || 1;
881
+ const level = Math.min(Math.max(Number(n.metadata?.level) || 1, 1), 6);
501
882
  return `<h${level}>${co}</h${level}>`;
502
883
  }
503
884
  case 'table': return await this.renderTableAsHtml(n);