officeparser 7.2.3 → 7.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +277 -17
- package/dist/OfficeConverter.d.ts +1 -1
- package/dist/OfficeConverter.js +3 -0
- package/dist/OfficeGenerator.js +4 -0
- package/dist/OfficeParser.d.ts +2 -0
- package/dist/OfficeParser.js +6 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +5 -2
- package/dist/defaults.js +12 -0
- package/dist/generators/BaseGenerator.d.ts +34 -1
- package/dist/generators/BaseGenerator.js +98 -0
- package/dist/generators/CsvGenerator.d.ts +9 -1
- package/dist/generators/CsvGenerator.js +28 -16
- package/dist/generators/EpubGenerator.d.ts +43 -0
- package/dist/generators/EpubGenerator.js +312 -0
- package/dist/generators/HtmlGenerator.d.ts +12 -0
- package/dist/generators/HtmlGenerator.js +378 -61
- package/dist/generators/MarkdownGenerator.d.ts +28 -5
- package/dist/generators/MarkdownGenerator.js +432 -51
- package/dist/generators/PdfGenerator.js +32 -0
- package/dist/generators/RtfGenerator.js +47 -22
- package/dist/generators/TextGenerator.js +98 -11
- package/dist/index.d.ts +1 -0
- package/dist/index.js +1 -0
- package/dist/officeparser.browser.d.ts +427 -20
- package/dist/officeparser.browser.iife.js +338 -206
- package/dist/officeparser.browser.mjs +346 -214
- package/dist/officeparser.browser.slim.d.ts +427 -20
- package/dist/officeparser.browser.slim.iife.js +346 -214
- package/dist/officeparser.browser.slim.mjs +346 -214
- package/dist/parsers/EpubParser.d.ts +8 -0
- package/dist/parsers/EpubParser.js +217 -0
- package/dist/parsers/ExcelParser.js +2 -0
- package/dist/parsers/HtmlParser.js +507 -48
- package/dist/parsers/MarkdownParser.js +704 -92
- package/dist/parsers/OpenOfficeParser.js +128 -20
- package/dist/parsers/PdfParser.js +4 -1
- package/dist/parsers/PowerPointParser.js +1 -0
- package/dist/parsers/WordParser.js +1 -0
- package/dist/sbom.cdx.json +1695 -0
- package/dist/types.d.ts +427 -20
- package/dist/types.js +8 -0
- package/dist/utils/configUtils.js +53 -4
- package/dist/utils/errorUtils.js +7 -3
- package/dist/utils/sanitize.d.ts +139 -0
- package/dist/utils/sanitize.js +318 -0
- package/dist/utils/xmlUtils.js +2 -2
- package/dist/utils/zipUtils.js +76 -26
- package/package.json +16 -12
|
@@ -1,7 +1,87 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
3
|
exports.MarkdownGenerator = void 0;
|
|
4
|
+
const sanitize_js_1 = require("../utils/sanitize.js");
|
|
4
5
|
const BaseGenerator_js_1 = require("./BaseGenerator.js");
|
|
6
|
+
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
7
|
+
/**
|
|
8
|
+
* Values accepted for an attribute-list `align=`. Matches what `MarkdownParser`'s own
|
|
9
|
+
* `parseAttributeList` allowlists on import (plus `justify`, which HTML sources can supply),
|
|
10
|
+
* so this is lossless for anything the parser produced.
|
|
11
|
+
*/
|
|
12
|
+
const MD_ALIGN_VALUES = new Set(['left', 'center', 'right', 'justify']);
|
|
13
|
+
/** A CSS length or percentage - the only shape `width=` legitimately carries. */
|
|
14
|
+
const MD_LENGTH_PATTERN = /^\d+(?:\.\d+)?(?:px|%|em|rem|pt|pc|in|cm|mm|ex|ch|vw|vh)?$/;
|
|
15
|
+
/** Admonition kinds, mirroring the union declared on `AdmonitionMetadata` in types.ts. */
|
|
16
|
+
const MD_ADMONITION_TYPES = new Set(['note', 'tip', 'important', 'warning', 'caution']);
|
|
17
|
+
/**
|
|
18
|
+
* Folds line breaks to spaces.
|
|
19
|
+
*
|
|
20
|
+
* Used on values that sit inside a single-line construct (an abbreviation definition, an
|
|
21
|
+
* admonition's bold title). A raw newline there does not merely look wrong: it terminates the
|
|
22
|
+
* construct and exposes whatever follows as document-level Markdown.
|
|
23
|
+
*/
|
|
24
|
+
const foldLines = (value) => String(value ?? '').replace(/[\r\n]+/g, ' ');
|
|
25
|
+
/**
|
|
26
|
+
* Named Markdown dialect presets. `extended` reproduces this library's historical output
|
|
27
|
+
* exactly (every feature on, GitHub-style admonitions) - the backward-compatibility anchor.
|
|
28
|
+
*/
|
|
29
|
+
const MARKDOWN_DIALECT_PRESETS = {
|
|
30
|
+
extended: { admonitions: 'github', definitionLists: true, footnotes: true, citations: true, wikilinks: true, math: 'dollar', attributeLists: true, strikethrough: true, bulletListMarker: '-', orderedListMarker: '.', emphasisMarker: 'asterisk', tables: 'native' },
|
|
31
|
+
github: { admonitions: 'github', definitionLists: false, footnotes: true, citations: false, wikilinks: false, math: 'dollar', attributeLists: false, strikethrough: true, bulletListMarker: '-', orderedListMarker: '.', emphasisMarker: 'asterisk', tables: 'native' },
|
|
32
|
+
gitlab: { admonitions: 'gitlab', definitionLists: false, footnotes: true, citations: false, wikilinks: false, math: 'dollar', attributeLists: false, strikethrough: true, bulletListMarker: '-', orderedListMarker: '.', emphasisMarker: 'asterisk', tables: 'native' },
|
|
33
|
+
obsidian: { admonitions: 'github', definitionLists: false, footnotes: true, citations: false, wikilinks: true, math: 'dollar', attributeLists: false, strikethrough: true, bulletListMarker: '-', orderedListMarker: '.', emphasisMarker: 'asterisk', tables: 'native' },
|
|
34
|
+
pandoc: { admonitions: 'pandoc', definitionLists: true, footnotes: true, citations: true, wikilinks: false, math: 'dollar', attributeLists: true, strikethrough: true, bulletListMarker: '-', orderedListMarker: '.', emphasisMarker: 'asterisk', tables: 'native' },
|
|
35
|
+
commonmark: { admonitions: 'none', definitionLists: false, footnotes: false, citations: false, wikilinks: false, math: 'none', attributeLists: false, strikethrough: false, bulletListMarker: '-', orderedListMarker: '.', emphasisMarker: 'asterisk', tables: 'html' },
|
|
36
|
+
};
|
|
37
|
+
/**
|
|
38
|
+
* Normalizes `MdGeneratorConfig.dialect` into a fully-resolved preset. A string names a preset
|
|
39
|
+
* directly; an object's `extends` field (default `'extended'`) names the base preset that any
|
|
40
|
+
* omitted field falls back to - NOT "whatever preset was ambient before", since config merging
|
|
41
|
+
* replaces the whole `dialect` field rather than layering an object on top of a prior string.
|
|
42
|
+
*/
|
|
43
|
+
function resolveDialect(dialect) {
|
|
44
|
+
if (dialect === undefined)
|
|
45
|
+
return MARKDOWN_DIALECT_PRESETS.extended;
|
|
46
|
+
if (typeof dialect === 'string')
|
|
47
|
+
return MARKDOWN_DIALECT_PRESETS[dialect] ?? MARKDOWN_DIALECT_PRESETS.extended;
|
|
48
|
+
const base = MARKDOWN_DIALECT_PRESETS[dialect.extends ?? 'extended'] ?? MARKDOWN_DIALECT_PRESETS.extended;
|
|
49
|
+
return {
|
|
50
|
+
admonitions: dialect.admonitions ?? base.admonitions,
|
|
51
|
+
definitionLists: dialect.definitionLists ?? base.definitionLists,
|
|
52
|
+
footnotes: dialect.footnotes ?? base.footnotes,
|
|
53
|
+
citations: dialect.citations ?? base.citations,
|
|
54
|
+
wikilinks: dialect.wikilinks ?? base.wikilinks,
|
|
55
|
+
math: dialect.math ?? base.math,
|
|
56
|
+
attributeLists: dialect.attributeLists ?? base.attributeLists,
|
|
57
|
+
strikethrough: dialect.strikethrough ?? base.strikethrough,
|
|
58
|
+
bulletListMarker: dialect.bulletListMarker ?? base.bulletListMarker,
|
|
59
|
+
orderedListMarker: dialect.orderedListMarker ?? base.orderedListMarker,
|
|
60
|
+
emphasisMarker: dialect.emphasisMarker ?? base.emphasisMarker,
|
|
61
|
+
tables: dialect.tables ?? base.tables,
|
|
62
|
+
};
|
|
63
|
+
}
|
|
64
|
+
/**
|
|
65
|
+
* Normalizes `MdGeneratorConfig.fallbackToHtml` into a fully resolved object, mirroring
|
|
66
|
+
* `HtmlGenerator`'s `resolveStandalone()` pattern: `true`/undefined turns every part on; `false`
|
|
67
|
+
* turns every part off; an object's omitted fields default to on.
|
|
68
|
+
*/
|
|
69
|
+
function resolveFallbackToHtml(fallbackToHtml) {
|
|
70
|
+
const uniform = (on) => ({
|
|
71
|
+
textFormatting: on, alignment: on, anchors: on, tables: on, embeds: on, cellLineBreaks: on,
|
|
72
|
+
});
|
|
73
|
+
if (fallbackToHtml === undefined || typeof fallbackToHtml === 'boolean')
|
|
74
|
+
return uniform(fallbackToHtml ?? true);
|
|
75
|
+
const on = uniform(true);
|
|
76
|
+
return {
|
|
77
|
+
textFormatting: fallbackToHtml.textFormatting ?? on.textFormatting,
|
|
78
|
+
alignment: fallbackToHtml.alignment ?? on.alignment,
|
|
79
|
+
anchors: fallbackToHtml.anchors ?? on.anchors,
|
|
80
|
+
tables: fallbackToHtml.tables ?? on.tables,
|
|
81
|
+
embeds: fallbackToHtml.embeds ?? on.embeds,
|
|
82
|
+
cellLineBreaks: fallbackToHtml.cellLineBreaks ?? on.cellLineBreaks,
|
|
83
|
+
};
|
|
84
|
+
}
|
|
5
85
|
/**
|
|
6
86
|
* Generates Markdown from an AST.
|
|
7
87
|
*
|
|
@@ -11,11 +91,11 @@ const BaseGenerator_js_1 = require("./BaseGenerator.js");
|
|
|
11
91
|
* be used for these features.
|
|
12
92
|
*
|
|
13
93
|
* 2. **Fidelity vs. Purity (The `fallbackToHtml` Principle)**:
|
|
14
|
-
* - When `fallbackToHtml` is TRUE: The generator prioritizes high-fidelity
|
|
15
|
-
* conversion. It will use HTML tags for features that Markdown
|
|
16
|
-
* represent (e.g., `<u>` for underline, `<div>` for alignment, `<table>`
|
|
17
|
-
* nested structures or merged cells).
|
|
18
|
-
* - When
|
|
94
|
+
* - When a given `fallbackToHtml` part is TRUE: The generator prioritizes high-fidelity
|
|
95
|
+
* document conversion for that part. It will use HTML tags for features that Markdown
|
|
96
|
+
* cannot natively represent (e.g., `<u>` for underline, `<div>` for alignment, `<table>`
|
|
97
|
+
* for nested structures or merged cells).
|
|
98
|
+
* - When FALSE: The generator prioritizes "pure" Markdown for that part.
|
|
19
99
|
* Unsupported features are either:
|
|
20
100
|
* - **Skipped**: Non-essential formatting like underline, subscript, superscript,
|
|
21
101
|
* or text alignment is omitted.
|
|
@@ -25,22 +105,82 @@ const BaseGenerator_js_1 = require("./BaseGenerator.js");
|
|
|
25
105
|
*
|
|
26
106
|
* 3. **Consistency**: All similar structural or formatting ideological problems must be
|
|
27
107
|
* resolved using these same rules to ensure predictable output.
|
|
108
|
+
*
|
|
109
|
+
* 4. **Dialect (`MdGeneratorConfig.dialect`)**: A second, independent axis from `fallbackToHtml` -
|
|
110
|
+
* which *native* Markdown syntax to emit for constructs with more than one real-world
|
|
111
|
+
* convention (admonitions, definition lists, footnotes, citations, wikilinks, math, list/
|
|
112
|
+
* emphasis markers, tables). See `resolveDialect()` and `MARKDOWN_DIALECT_PRESETS` above.
|
|
28
113
|
*/
|
|
29
114
|
class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
30
115
|
isInsideTable = false;
|
|
31
116
|
hoistedContent = [];
|
|
117
|
+
collectedAbbreviations = new Map();
|
|
118
|
+
resolvedDialect;
|
|
119
|
+
resolvedFallbackToHtml;
|
|
32
120
|
constructor(ast, config) {
|
|
33
121
|
super('md', ast, config);
|
|
122
|
+
this.resolvedDialect = resolveDialect(this.config.mdConfig.dialect);
|
|
123
|
+
this.resolvedFallbackToHtml = resolveFallbackToHtml(this.config.mdConfig.fallbackToHtml);
|
|
34
124
|
}
|
|
35
125
|
/**
|
|
36
126
|
* Renders anchor tags if HTML fallback is allowed.
|
|
37
127
|
*/
|
|
38
128
|
renderAnchors(metadata) {
|
|
39
|
-
if (!this.
|
|
129
|
+
if (!this.resolvedFallbackToHtml.anchors || this.config.ignoreInternalLinks)
|
|
40
130
|
return '';
|
|
41
131
|
const ids = metadata?.anchorIds || [];
|
|
42
132
|
return ids.map((aid) => `<a id="${this.slugify(aid)}"></a>`).join('');
|
|
43
133
|
}
|
|
134
|
+
/**
|
|
135
|
+
* Serializes a frontmatter array as a YAML flow sequence (e.g. `[a, b]`), matching
|
|
136
|
+
* MarkdownParser's frontmatter array handling. Plain strings are left bare; anything
|
|
137
|
+
* that would break flow-array syntax (or isn't a string) falls back to JSON encoding.
|
|
138
|
+
*/
|
|
139
|
+
serializeFrontmatterArray(arr) {
|
|
140
|
+
const items = arr.map(item => (typeof item === 'string' && item.trim() === item && !/[,[\]]/.test(item))
|
|
141
|
+
? item
|
|
142
|
+
: JSON.stringify(item));
|
|
143
|
+
return `[${items.join(', ')}]`;
|
|
144
|
+
}
|
|
145
|
+
/**
|
|
146
|
+
* Renders a Pandoc-style attribute list (e.g. `{width=50% align=left}`) from
|
|
147
|
+
* ImageMetadata/TableMetadata's width/align fields - the canonical form is always
|
|
148
|
+
* `key=value`, matching MarkdownParser's own vocabulary (MARKDOWN_DIALECT.md §15).
|
|
149
|
+
*/
|
|
150
|
+
renderAttributeList(meta) {
|
|
151
|
+
if (!this.resolvedDialect.attributeLists)
|
|
152
|
+
return '';
|
|
153
|
+
if (!meta?.width && !meta?.align)
|
|
154
|
+
return '';
|
|
155
|
+
const parts = [];
|
|
156
|
+
// Allowlist, not escape. These land in `metadata.width`/`align` on reparse, which the
|
|
157
|
+
// parser does NOT entity-decode, so encoding here would not round-trip - and stripping
|
|
158
|
+
// alone is not enough: the previous `[{}\s]+` guard removed whitespace, which stops
|
|
159
|
+
// `<img src=x onerror=…>` but not the slash-separated `<img/src=x/onerror=…>`.
|
|
160
|
+
// Both values have a small, fully-known shape, so matching that shape is both safer and
|
|
161
|
+
// lossless for anything a parser can produce.
|
|
162
|
+
//
|
|
163
|
+
// (`isValidContainerWidth` in utils/configUtils.ts is a near-identical regex, but it is a
|
|
164
|
+
// config validator that also accepts 'auto' and numbers; importing configUtils here for
|
|
165
|
+
// one pattern would be a worse coupling than this local constant.)
|
|
166
|
+
if (meta.width && MD_LENGTH_PATTERN.test(String(meta.width).trim())) {
|
|
167
|
+
parts.push(`width=${String(meta.width).trim()}`);
|
|
168
|
+
}
|
|
169
|
+
if (meta.align && MD_ALIGN_VALUES.has(String(meta.align).trim().toLowerCase())) {
|
|
170
|
+
parts.push(`align=${String(meta.align).trim().toLowerCase()}`);
|
|
171
|
+
}
|
|
172
|
+
if (parts.length === 0)
|
|
173
|
+
return '';
|
|
174
|
+
return `{${parts.join(' ')}}`;
|
|
175
|
+
}
|
|
176
|
+
/** Converts a document-supplied date to an ISO string, or '' if invalid
|
|
177
|
+
* (a malformed date would otherwise throw a RangeError and abort generation). */
|
|
178
|
+
toIsoDate(value) {
|
|
179
|
+
if (value === undefined || value === null || value === '')
|
|
180
|
+
return '';
|
|
181
|
+
const d = new Date(value);
|
|
182
|
+
return isNaN(d.getTime()) ? '' : d.toISOString();
|
|
183
|
+
}
|
|
44
184
|
/**
|
|
45
185
|
* Generates Markdown string from the provided AST.
|
|
46
186
|
*
|
|
@@ -49,21 +189,34 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
49
189
|
async generate() {
|
|
50
190
|
let output = '';
|
|
51
191
|
// Add Metadata (YAML Front Matter)
|
|
52
|
-
|
|
192
|
+
const meta = this.effectiveMetadata;
|
|
193
|
+
if (meta) {
|
|
53
194
|
output += '---\n';
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
if (
|
|
59
|
-
output += `
|
|
60
|
-
if (
|
|
61
|
-
output += `
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
195
|
+
// JSON-encode scalar values so a title/author/description containing a
|
|
196
|
+
// quote or newline can't break out of the YAML string and inject
|
|
197
|
+
// arbitrary front-matter keys. (JSON.stringify of a benign value yields
|
|
198
|
+
// the same `"..."` form as before, so normal output is unchanged.)
|
|
199
|
+
if (meta.title)
|
|
200
|
+
output += `title: ${JSON.stringify(meta.title)}\n`;
|
|
201
|
+
if (meta.author)
|
|
202
|
+
output += `author: ${JSON.stringify(meta.author)}\n`;
|
|
203
|
+
const createdIso = this.toIsoDate(meta.created);
|
|
204
|
+
if (createdIso)
|
|
205
|
+
output += `created: ${createdIso}\n`;
|
|
206
|
+
const modifiedIso = this.toIsoDate(meta.modified);
|
|
207
|
+
if (modifiedIso)
|
|
208
|
+
output += `modified: ${modifiedIso}\n`;
|
|
209
|
+
if (meta.description)
|
|
210
|
+
output += `description: ${JSON.stringify(meta.description)}\n`;
|
|
211
|
+
if (meta.subject)
|
|
212
|
+
output += `subject: ${JSON.stringify(meta.subject)}\n`;
|
|
213
|
+
if (meta.keywords)
|
|
214
|
+
output += `keywords: ${JSON.stringify(meta.keywords)}\n`;
|
|
215
|
+
if (meta.customProperties) {
|
|
216
|
+
for (const [key, val] of Object.entries(meta.customProperties)) {
|
|
217
|
+
// Strip newlines/colons from the key so it can't inject a new mapping.
|
|
218
|
+
const safeKey = String(key).replace(/[\r\n:]+/g, ' ').trim();
|
|
219
|
+
output += `${safeKey}: ${Array.isArray(val) ? this.serializeFrontmatterArray(val) : JSON.stringify(val)}\n`;
|
|
67
220
|
}
|
|
68
221
|
}
|
|
69
222
|
output += '---\n\n';
|
|
@@ -87,16 +240,19 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
87
240
|
}
|
|
88
241
|
switch (node.type) {
|
|
89
242
|
case 'text': {
|
|
90
|
-
|
|
243
|
+
// Entity-encode angle brackets so document text can't inject a raw
|
|
244
|
+
// HTML tag (e.g. <script>) when the Markdown is rendered to HTML.
|
|
245
|
+
let text = (0, sanitize_js_1.markdownEscapeText)(node.text || '');
|
|
91
246
|
if (this.config.includeFormatting && node.formatting) {
|
|
247
|
+
const emphasisAsterisk = this.resolvedDialect.emphasisMarker === 'asterisk';
|
|
92
248
|
if (node.formatting.bold)
|
|
93
|
-
text = `**${text}
|
|
249
|
+
text = emphasisAsterisk ? `**${text}**` : `__${text}__`;
|
|
94
250
|
if (node.formatting.italic)
|
|
95
|
-
text = `*${text}
|
|
96
|
-
if (node.formatting.strikethrough)
|
|
251
|
+
text = emphasisAsterisk ? `*${text}*` : `_${text}_`;
|
|
252
|
+
if (node.formatting.strikethrough && this.resolvedDialect.strikethrough)
|
|
97
253
|
text = `~~${text}~~`;
|
|
98
254
|
// Use HTML tags for formatting not natively supported by standard Markdown
|
|
99
|
-
if (this.
|
|
255
|
+
if (this.resolvedFallbackToHtml.textFormatting) {
|
|
100
256
|
if (node.formatting.underline)
|
|
101
257
|
text = `<u>${text}</u>`;
|
|
102
258
|
if (node.formatting.subscript)
|
|
@@ -106,18 +262,52 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
106
262
|
}
|
|
107
263
|
}
|
|
108
264
|
const meta = node.metadata;
|
|
109
|
-
if (meta?.
|
|
265
|
+
if (meta?.wikilink && this.resolvedDialect.wikilinks) {
|
|
266
|
+
// Obsidian syntax: bare page name, or page|alias when the display
|
|
267
|
+
// text differs from the page name. Strip the `[]|`/newline chars
|
|
268
|
+
// that would break out of the `[[...]]` wrapper.
|
|
269
|
+
// The alias must be built from the ESCAPED text, not from raw node.text.
|
|
270
|
+
// Rebuilding from the raw value here discarded the markdownEscapeText()
|
|
271
|
+
// applied above, so a wikilink was the one place document text reached
|
|
272
|
+
// the output unescaped. Escaping is lossless for the alias specifically,
|
|
273
|
+
// because it lands back in a text node, which the parser entity-decodes.
|
|
274
|
+
const alias = (0, sanitize_js_1.markdownEscapeText)(node.text || '').replace(/[[\]|\r\n]+/g, '');
|
|
275
|
+
// `page` lands in metadata.link, which is NOT entity-decoded on reparse,
|
|
276
|
+
// so it gets `<` dropped rather than encoded - a page name is an
|
|
277
|
+
// identifier, and `<` carries no meaning in one.
|
|
278
|
+
const page = (meta.link || '').replace(/[[\]|<\r\n]+/g, '');
|
|
279
|
+
text = (node.text && node.text !== (meta.link || '')) ? `[[${page}|${alias}]]` : `[[${page}]]`;
|
|
280
|
+
}
|
|
281
|
+
else if (meta?.link) {
|
|
110
282
|
const isInternal = meta.linkType !== 'external';
|
|
111
283
|
if (!this.config.ignoreInternalLinks || !isInternal) {
|
|
112
284
|
let link = meta.link;
|
|
113
285
|
// Slugify internal link targets to match heading IDs if generating IDs
|
|
114
|
-
if (isInternal && link.startsWith('#') && (this.config.generateIds || this.
|
|
286
|
+
if (isInternal && link.startsWith('#') && (this.config.generateIds || this.resolvedFallbackToHtml.anchors)) {
|
|
115
287
|
const target = link.substring(1);
|
|
116
288
|
link = '#' + this.slugify(target);
|
|
117
289
|
}
|
|
118
|
-
|
|
290
|
+
// Reject javascript:/data: schemes and encode `()`/whitespace so the
|
|
291
|
+
// URL can't break out of `](...)` or inject a script link.
|
|
292
|
+
text = `[${text}](${(0, sanitize_js_1.sanitizeMarkdownUrl)(link)})`;
|
|
119
293
|
}
|
|
120
294
|
}
|
|
295
|
+
if (meta?.abbreviationTitle) {
|
|
296
|
+
// Markdown Extra's abbreviation syntax has no inline marker - the bare
|
|
297
|
+
// word round-trips as-is, with its expansion collected at the document
|
|
298
|
+
// end via `*[abbr]: title`.
|
|
299
|
+
this.collectedAbbreviations.set(node.text || '', meta.abbreviationTitle);
|
|
300
|
+
}
|
|
301
|
+
if (meta?.citationKey) {
|
|
302
|
+
// Allowlist to exactly the character class MarkdownParser's own citation
|
|
303
|
+
// recognizer accepts, so this is provably lossless for anything it
|
|
304
|
+
// produced - while fully neutralizing a key arriving from HtmlParser's
|
|
305
|
+
// `data-citation-key`, which accepts any string. Like the wikilink above,
|
|
306
|
+
// this branch also replaces `text` wholesale, so a strip that left `<`
|
|
307
|
+
// behind discarded the escaping applied earlier.
|
|
308
|
+
const key = String(meta.citationKey).replace(/[^a-zA-Z0-9_:.-]/g, '');
|
|
309
|
+
text = this.resolvedDialect.citations ? `[@${key}]` : `[${key}]`;
|
|
310
|
+
}
|
|
121
311
|
return text;
|
|
122
312
|
}
|
|
123
313
|
case 'heading': {
|
|
@@ -136,14 +326,14 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
136
326
|
else if (this.config.generateIds) {
|
|
137
327
|
id = ` {#${this.slugify(this.getNodeText(node))}}`;
|
|
138
328
|
}
|
|
139
|
-
const anchors = this.
|
|
140
|
-
? remainingAnchors.map(aid => `<a name="${aid}"></a>`).join('')
|
|
329
|
+
const anchors = this.resolvedFallbackToHtml.anchors
|
|
330
|
+
? remainingAnchors.map(aid => `<a name="${this.slugify(aid)}"></a>`).join('')
|
|
141
331
|
: '';
|
|
142
332
|
let content = `${prefix}${childrenOutput}${id}`;
|
|
143
333
|
// Alignment fallback via HTML div/p
|
|
144
|
-
if (this.
|
|
334
|
+
if (this.resolvedFallbackToHtml.alignment && meta?.alignment && meta.alignment !== 'left') {
|
|
145
335
|
// Use extra newlines to ensure Markdown inside the div is parsed
|
|
146
|
-
content = `<div style="text-align: ${meta.alignment}">\n\n${content}\n\n</div>`;
|
|
336
|
+
content = `<div style="text-align: ${(0, sanitize_js_1.sanitizeCssValue)(meta.alignment)}">\n\n${content}\n\n</div>`;
|
|
147
337
|
}
|
|
148
338
|
return `${anchors}${anchors ? '\n' : ''}${content}\n\n`;
|
|
149
339
|
}
|
|
@@ -152,8 +342,8 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
152
342
|
const anchors = this.renderAnchors(meta);
|
|
153
343
|
let content = childrenOutput;
|
|
154
344
|
// Alignment fallback via HTML div/p
|
|
155
|
-
if (this.
|
|
156
|
-
content = `<div style="text-align: ${meta.alignment}">${content}</div>`;
|
|
345
|
+
if (this.resolvedFallbackToHtml.alignment && meta?.alignment && meta.alignment !== 'left') {
|
|
346
|
+
content = `<div style="text-align: ${(0, sanitize_js_1.sanitizeCssValue)(meta.alignment)}">${content}</div>`;
|
|
157
347
|
}
|
|
158
348
|
return childrenOutput ? `${anchors}${content}\n\n` : '';
|
|
159
349
|
}
|
|
@@ -161,7 +351,10 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
161
351
|
const meta = node.metadata;
|
|
162
352
|
const indentSpaces = ' '.repeat(4);
|
|
163
353
|
const indent = indentSpaces.repeat(meta?.indentation || 0);
|
|
164
|
-
const
|
|
354
|
+
const bullet = `${this.resolvedDialect.bulletListMarker} `;
|
|
355
|
+
const marker = meta?.isTask
|
|
356
|
+
? (meta.checked ? `${bullet}[x] ` : `${bullet}[ ] `)
|
|
357
|
+
: (meta?.listType === 'ordered' ? `${(meta.itemIndex ?? 0) + 1}${this.resolvedDialect.orderedListMarker} ` : bullet);
|
|
165
358
|
const anchors = this.renderAnchors(meta);
|
|
166
359
|
return `${indent}${marker}${anchors}${childrenOutput}\n`;
|
|
167
360
|
}
|
|
@@ -179,11 +372,25 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
179
372
|
}
|
|
180
373
|
}
|
|
181
374
|
const anchors = this.renderAnchors(meta);
|
|
182
|
-
|
|
375
|
+
// Strip `[]` from alt (would close the `![...]`) and neutralize the URL scheme.
|
|
376
|
+
const safeAlt = (0, sanitize_js_1.markdownEscapeText)(alt).replace(/[[\]]/g, '');
|
|
377
|
+
const safeSrc = (0, sanitize_js_1.sanitizeMarkdownUrl)(src, { allowDataImage: true });
|
|
378
|
+
return `${anchors}${anchors ? '\n' : ''}${this.renderAttributeList(meta)}`;
|
|
183
379
|
}
|
|
184
380
|
case 'table': {
|
|
185
381
|
const anchors = this.renderAnchors(node.metadata);
|
|
186
382
|
const tableOutput = await this.renderMarkdownTable(node, processor);
|
|
383
|
+
// The HTML-fallback path (merged cells/nested tables, or a dialect that forces
|
|
384
|
+
// HTML tables outright) already carries data-align on the <table> tag directly -
|
|
385
|
+
// only the plain pipe-table form needs the attribute-list syntax for alignment.
|
|
386
|
+
const usedHtmlFallback = this.resolvedDialect.tables === 'html' ||
|
|
387
|
+
(this.resolvedFallbackToHtml.tables && (this.hasNestedTable(node) || this.hasColspanOrRowspan(node)));
|
|
388
|
+
const attrList = usedHtmlFallback ? '' : this.renderAttributeList(node.metadata);
|
|
389
|
+
if (attrList) {
|
|
390
|
+
// Must glue directly below the last row with no blank line, or
|
|
391
|
+
// MarkdownParser's block splitter won't see it as part of the same block.
|
|
392
|
+
return `${anchors}${anchors ? '\n' : ''}${tableOutput.replace(/\n+$/, '\n')}${attrList}\n`;
|
|
393
|
+
}
|
|
187
394
|
return `${anchors}${anchors ? '\n' : ''}${tableOutput}`;
|
|
188
395
|
}
|
|
189
396
|
case 'row':
|
|
@@ -192,17 +399,60 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
192
399
|
return childrenOutput;
|
|
193
400
|
}
|
|
194
401
|
case 'break': {
|
|
402
|
+
// A hard line break (CommonMark: two trailing spaces before the
|
|
403
|
+
// newline) round-trips back to a distinct 'break' node on reparse;
|
|
404
|
+
// every other breakType (including 'page', used for a thematic-break
|
|
405
|
+
// HR) keeps emitting a bare newline, unchanged.
|
|
406
|
+
const meta = node.metadata;
|
|
407
|
+
if (meta?.breakType === 'carriageReturn')
|
|
408
|
+
return ' \n';
|
|
195
409
|
return '\n';
|
|
196
410
|
}
|
|
197
411
|
case 'code': {
|
|
198
412
|
const meta = node.metadata;
|
|
199
|
-
|
|
200
|
-
//
|
|
201
|
-
|
|
202
|
-
|
|
413
|
+
// Math content reached the output completely raw, which mattered most under
|
|
414
|
+
// `math: 'none'` (the commonmark preset), where there is no `$` wrapper at
|
|
415
|
+
// all and the text lands directly in the document body.
|
|
416
|
+
//
|
|
417
|
+
// Encode rather than drop: `$a < b$` is ordinary LaTeX, and dropping `<`
|
|
418
|
+
// would silently corrupt real formulae. markdownEscapeText only touches `<`
|
|
419
|
+
// followed by a letter/`/`/`!`/`?`, which is not idiomatic math, and it is
|
|
420
|
+
// idempotent - so output is stable across repeated round-trips even though
|
|
421
|
+
// the first cycle shifts an anomalous `<img` to `<img`. (Fully lossless
|
|
422
|
+
// would mean teaching MarkdownParser.decodeHtmlEntities to cover math `code`
|
|
423
|
+
// nodes; that is a parser behaviour change with its own baseline
|
|
424
|
+
// consequences and must not gate a security fix.)
|
|
425
|
+
if (meta?.math === 'block') {
|
|
426
|
+
// A content line of exactly `$$` would close the block early.
|
|
427
|
+
const mathBlock = (0, sanitize_js_1.markdownEscapeText)(node.text || '')
|
|
428
|
+
.split('\n').map(l => (l.trim() === '$$' ? ` ${l}` : l)).join('\n');
|
|
429
|
+
return this.resolvedDialect.math === 'dollar' ? `\n$$\n${mathBlock}\n$$\n\n` : `\n${mathBlock}\n\n`;
|
|
430
|
+
}
|
|
431
|
+
if (meta?.math === 'inline') {
|
|
432
|
+
// Dropping `$` and newlines is lossless here: the parser's own inline-math
|
|
433
|
+
// recognizer is `\$(?!\s)([^$\n]+?)(?<!\s)\$`, which can never capture either.
|
|
434
|
+
const mathInline = (0, sanitize_js_1.markdownEscapeText)(node.text || '').replace(/[$\r\n]+/g, '');
|
|
435
|
+
return this.resolvedDialect.math === 'dollar' ? `$${mathInline}$` : mathInline;
|
|
436
|
+
}
|
|
437
|
+
const lang = (meta?.language || '').replace(/[\r\n`]+/g, '');
|
|
438
|
+
// Block code if it contains a line break, else inline. Testing only for `\n`
|
|
439
|
+
// routed a CR-only string to the inline branch, where a renderer that
|
|
440
|
+
// normalizes `\r` to a line ending sees a blank line, the span dies, and the
|
|
441
|
+
// remainder is exposed as raw Markdown. The fence sizing below is correct and
|
|
442
|
+
// needs no change; code content itself is not an HTML context.
|
|
443
|
+
if (node.text && /[\r\n]/.test(node.text)) {
|
|
444
|
+
// Fence with one more backtick than the longest run inside the content
|
|
445
|
+
// so an embedded ``` can't close the block early and inject markup.
|
|
446
|
+
const longestRun = Math.max(0, ...(node.text.match(/`+/g) || []).map(s => s.length));
|
|
447
|
+
const fence = '`'.repeat(Math.max(3, longestRun + 1));
|
|
448
|
+
return `\n${fence}${lang}\n${node.text}\n${fence}\n\n`;
|
|
203
449
|
}
|
|
204
450
|
else {
|
|
205
|
-
|
|
451
|
+
const t = node.text || '';
|
|
452
|
+
const longestRun = Math.max(0, ...(t.match(/`+/g) || []).map(s => s.length));
|
|
453
|
+
const fence = '`'.repeat(Math.max(1, longestRun + 1));
|
|
454
|
+
const pad = (t.startsWith('`') || t.endsWith('`')) ? ' ' : '';
|
|
455
|
+
return `${fence}${pad}${t}${pad}${fence} `;
|
|
206
456
|
}
|
|
207
457
|
}
|
|
208
458
|
case 'sheet': {
|
|
@@ -220,9 +470,80 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
220
470
|
}
|
|
221
471
|
case 'note': {
|
|
222
472
|
const meta = node.metadata;
|
|
223
|
-
|
|
224
|
-
|
|
473
|
+
if (meta?.noteType === 'footnote' || meta?.noteType === 'endnote') {
|
|
474
|
+
if (!this.resolvedDialect.footnotes) {
|
|
475
|
+
// Dialect has no footnote syntax - the caller inlines this bare body
|
|
476
|
+
// as a parenthetical at the reference point instead of collecting it
|
|
477
|
+
// into an end-of-document "### Notes" section under a [^id] marker.
|
|
478
|
+
return childrenOutput.trim();
|
|
479
|
+
}
|
|
480
|
+
return `[^${this.getFootnoteKey(node)}]: ${childrenOutput.trim()}\n\n`;
|
|
481
|
+
}
|
|
482
|
+
return `> **Note:** ${childrenOutput.trim()}\n\n`;
|
|
225
483
|
}
|
|
484
|
+
case 'embed': {
|
|
485
|
+
// Markdown has no native embed syntax. When fallbackToHtml.embeds is on (our
|
|
486
|
+
// save default), emit the exact single-line div MarkdownParser recognises on
|
|
487
|
+
// reimport; otherwise degrade to a plain link.
|
|
488
|
+
const meta = node.metadata;
|
|
489
|
+
const id = meta?.videoId || '';
|
|
490
|
+
if (this.resolvedFallbackToHtml.embeds) {
|
|
491
|
+
const width = meta?.width ? ` data-width="${(0, sanitize_js_1.escapeHtml)(meta.width)}"` : '';
|
|
492
|
+
const align = meta?.align ? ` data-align="${(0, sanitize_js_1.escapeHtml)(meta.align)}"` : '';
|
|
493
|
+
return `\n<div data-youtube-video="${(0, sanitize_js_1.escapeHtml)(id)}"${width}${align}></div>\n\n`;
|
|
494
|
+
}
|
|
495
|
+
const url = meta?.url || (id ? `https://youtu.be/${id}` : '');
|
|
496
|
+
return url ? `[YouTube](${(0, sanitize_js_1.sanitizeMarkdownUrl)(url)})\n\n` : '';
|
|
497
|
+
}
|
|
498
|
+
case 'admonition': {
|
|
499
|
+
const meta = node.metadata;
|
|
500
|
+
// `admonitionType` is a closed union in types.ts and both parsers already
|
|
501
|
+
// allowlist on import, so enforcing it here is a no-op for any conforming
|
|
502
|
+
// AST - it closes the gap for a programmatically-built one, where the type is
|
|
503
|
+
// interpolated straight into `:::TYPE` / `::: {.TYPE}` / `[!TYPE]`.
|
|
504
|
+
const rawType = String(meta?.admonitionType || 'note').toLowerCase();
|
|
505
|
+
const type = MD_ADMONITION_TYPES.has(rawType) ? rawType : 'note';
|
|
506
|
+
const label = type.toUpperCase();
|
|
507
|
+
// A newline in the title would close the `**...**` and, in the fenced-div
|
|
508
|
+
// branches, could emit a stray `:::` line. `title` is never parser-set, so
|
|
509
|
+
// there is no round-trip to preserve and escaping is free.
|
|
510
|
+
const title = meta?.title ? (0, sanitize_js_1.markdownEscapeText)(foldLines(meta.title)) : '';
|
|
511
|
+
const body = childrenOutput.trim();
|
|
512
|
+
switch (this.resolvedDialect.admonitions) {
|
|
513
|
+
case 'gitlab':
|
|
514
|
+
// GLFM fenced-div: no dedicated title syntax, so a custom title (if
|
|
515
|
+
// any) is folded into the body as a bold first line.
|
|
516
|
+
return `:::${type}\n${title ? `**${title}**\n\n` : ''}${body}\n:::\n\n`;
|
|
517
|
+
case 'pandoc':
|
|
518
|
+
// Pandoc's own fenced-div-with-class syntax; same title handling as gitlab.
|
|
519
|
+
return `::: {.${type}}\n${title ? `**${title}**\n\n` : ''}${body}\n:::\n\n`;
|
|
520
|
+
case 'none': {
|
|
521
|
+
// Degrade to a plain bold-labeled blockquote, no special marker.
|
|
522
|
+
const quotedLines = body.split('\n').map(l => l.length > 0 ? `> ${l}` : '>').join('\n');
|
|
523
|
+
const heading = title || label.charAt(0) + label.slice(1).toLowerCase();
|
|
524
|
+
return `> **${heading}:**\n${quotedLines}\n\n`;
|
|
525
|
+
}
|
|
526
|
+
case 'github':
|
|
527
|
+
default: {
|
|
528
|
+
// Canonical GitHub blockquote form. No dedicated title syntax either
|
|
529
|
+
// (matches this library's historical output).
|
|
530
|
+
const quotedLines = body.split('\n').map(l => l.length > 0 ? `> ${l}` : '>').join('\n');
|
|
531
|
+
return `> [!${label}]\n${quotedLines}\n\n`;
|
|
532
|
+
}
|
|
533
|
+
}
|
|
534
|
+
}
|
|
535
|
+
case 'definitionList':
|
|
536
|
+
if (!this.resolvedDialect.definitionLists)
|
|
537
|
+
return `${childrenOutput}\n`;
|
|
538
|
+
return `${childrenOutput}\n`;
|
|
539
|
+
case 'definitionTerm':
|
|
540
|
+
if (!this.resolvedDialect.definitionLists)
|
|
541
|
+
return `**${childrenOutput}**\n\n`;
|
|
542
|
+
return `${childrenOutput}\n`;
|
|
543
|
+
case 'definitionDescription':
|
|
544
|
+
if (!this.resolvedDialect.definitionLists)
|
|
545
|
+
return `${childrenOutput}\n\n`;
|
|
546
|
+
return `: ${childrenOutput}\n`;
|
|
226
547
|
default:
|
|
227
548
|
return childrenOutput;
|
|
228
549
|
}
|
|
@@ -253,8 +574,24 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
253
574
|
}
|
|
254
575
|
output += notesMd;
|
|
255
576
|
}
|
|
577
|
+
if (this.collectedAbbreviations.size > 0) {
|
|
578
|
+
output += '\n\n';
|
|
579
|
+
for (const [abbr, title] of this.collectedAbbreviations) {
|
|
580
|
+
output += `*[${(0, sanitize_js_1.markdownEscapeText)(String(abbr).replace(/[[\]\r\n]+/g, ''))}]: ${(0, sanitize_js_1.markdownEscapeText)(foldLines(title))}\n`;
|
|
581
|
+
}
|
|
582
|
+
}
|
|
583
|
+
// Only a run of literal "\n" at either end is ever a generator artifact here: block
|
|
584
|
+
// separators, the notes/abbreviations sections, the unconditional '\n\n' before
|
|
585
|
+
// hoistedContent (added even when hoistedContent is empty), and renderMarkdownTable's
|
|
586
|
+
// HTML-fallback branches, which unconditionally wrap in a leading+trailing '\n' as
|
|
587
|
+
// separators from whatever precedes/follows (in practice this rarely surfaces at the very
|
|
588
|
+
// start of `output` today since frontmatter's own "---" almost always precedes real
|
|
589
|
+
// content first - see the type doc on `ast.metadata` - but the strip is correct regardless
|
|
590
|
+
// of what precedes it). Nothing else at either end is a generator artifact: not leading
|
|
591
|
+
// whitespace, and not any other kind of trailing whitespace, both of which would be real
|
|
592
|
+
// document content. See the identical reasoning in TextGenerator.generate().
|
|
256
593
|
return {
|
|
257
|
-
value: (output + '\n\n' + this.hoistedContent.join('\n\n')).
|
|
594
|
+
value: (output + '\n\n' + this.hoistedContent.join('\n\n')).replace(/^\n+|\n+$/g, ''),
|
|
258
595
|
messages: this.messages
|
|
259
596
|
};
|
|
260
597
|
}
|
|
@@ -263,6 +600,10 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
263
600
|
* Overridden to provide AST optimization (merging adjacent text nodes).
|
|
264
601
|
*/
|
|
265
602
|
async processNodeRecursive(node, processor) {
|
|
603
|
+
// Mirrors the check in BaseGenerator.processNodeRecursive. This override replaces that
|
|
604
|
+
// method entirely, so without repeating the check here the signal would be silently
|
|
605
|
+
// inert for this generator - which is exactly how it was missed.
|
|
606
|
+
(0, errorUtils_js_1.checkAbortSignal)(this.config.abortSignal);
|
|
266
607
|
// Allow user to completely override rendering or skip via onNode
|
|
267
608
|
const override = await this.handleOnNode(node);
|
|
268
609
|
if (override === false) {
|
|
@@ -279,9 +620,16 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
279
620
|
childrenOutput += await this.processNodeRecursive(child, processor);
|
|
280
621
|
}
|
|
281
622
|
}
|
|
623
|
+
// When the dialect has no footnote syntax, a footnote/endnote is inlined right at its
|
|
624
|
+
// reference point instead (see below) - so it must not also be collected into the
|
|
625
|
+
// end-of-document "### Notes" section, or its content would be duplicated.
|
|
626
|
+
const isInlinedFootnote = (note) => {
|
|
627
|
+
const meta = note.metadata;
|
|
628
|
+
return (meta?.noteType === 'footnote' || meta?.noteType === 'endnote') && !this.resolvedDialect.footnotes;
|
|
629
|
+
};
|
|
282
630
|
if (node.notes && node.notes.length > 0) {
|
|
283
631
|
if (node.type !== 'slide') {
|
|
284
|
-
this.collectedNotes.push(...node.notes);
|
|
632
|
+
this.collectedNotes.push(...node.notes.filter(note => !isInlinedFootnote(note)));
|
|
285
633
|
}
|
|
286
634
|
}
|
|
287
635
|
let result = await processor(node, childrenOutput);
|
|
@@ -290,6 +638,27 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
290
638
|
result += await this.processNodeRecursive(note, processor);
|
|
291
639
|
}
|
|
292
640
|
}
|
|
641
|
+
else if (node.notes && node.notes.length > 0) {
|
|
642
|
+
for (const note of node.notes) {
|
|
643
|
+
const meta = note.metadata;
|
|
644
|
+
if (meta?.noteType !== 'footnote' && meta?.noteType !== 'endnote')
|
|
645
|
+
continue;
|
|
646
|
+
if (isInlinedFootnote(note)) {
|
|
647
|
+
// Markdown-specific degrade (not RTF/plain-text's "drop the marker, just
|
|
648
|
+
// append at the end" convention): inline the note's rendered body as a
|
|
649
|
+
// parenthetical right where it's referenced, since Markdown readers benefit
|
|
650
|
+
// from an inline association those simpler formats don't need in the same way.
|
|
651
|
+
const body = await this.processNodeRecursive(note, processor);
|
|
652
|
+
result += ` (Note: ${body})`;
|
|
653
|
+
}
|
|
654
|
+
else {
|
|
655
|
+
// Emit the [^id] reference marker at the point of reference. Without this,
|
|
656
|
+
// a footnote/endnote would only ever show up in the collected ### Notes
|
|
657
|
+
// section at the end, with no indication of where it was originally cited.
|
|
658
|
+
result += `[^${this.getFootnoteKey(note)}]`;
|
|
659
|
+
}
|
|
660
|
+
}
|
|
661
|
+
}
|
|
293
662
|
return result;
|
|
294
663
|
}
|
|
295
664
|
/**
|
|
@@ -337,13 +706,19 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
337
706
|
async renderMarkdownTable(node, processor) {
|
|
338
707
|
if (!node.children || node.children.length === 0)
|
|
339
708
|
return '';
|
|
709
|
+
// A dialect that has no native table syntax at all (e.g. strict CommonMark) always
|
|
710
|
+
// renders as HTML, regardless of complexity - this is a separate axis from the
|
|
711
|
+
// nested/merged-cell HTML fallback below, which only applies to otherwise-native tables.
|
|
712
|
+
if (this.resolvedDialect.tables === 'html') {
|
|
713
|
+
return '\n' + await this.renderTableAsHtml(node) + '\n';
|
|
714
|
+
}
|
|
340
715
|
// If table is complex, nested, or uses merges, fallback to HTML for high fidelity if allowed
|
|
341
716
|
const isComplex = this.hasNestedTable(node) || this.hasColspanOrRowspan(node);
|
|
342
|
-
if (this.
|
|
717
|
+
if (this.resolvedFallbackToHtml.tables && isComplex) {
|
|
343
718
|
return '\n' + await this.renderTableAsHtml(node) + '\n';
|
|
344
719
|
}
|
|
345
720
|
// Handle nested tables in pure Markdown by hoisting them out
|
|
346
|
-
if (this.isInsideTable && !this.
|
|
721
|
+
if (this.isInsideTable && !this.resolvedFallbackToHtml.tables) {
|
|
347
722
|
const wasInside = this.isInsideTable;
|
|
348
723
|
this.isInsideTable = false; // Reset to allow rendering the hoisted table correctly
|
|
349
724
|
const hoistedId = this.hoistedContent.length + 1;
|
|
@@ -384,7 +759,7 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
384
759
|
// Process cell content
|
|
385
760
|
let cellContent = await this.processNodeRecursive(cellNode, processor);
|
|
386
761
|
// Use <br> fallback only if allowed, otherwise space
|
|
387
|
-
const br = this.
|
|
762
|
+
const br = this.resolvedFallbackToHtml.cellLineBreaks ? '<br>' : ' ';
|
|
388
763
|
cellContent = cellContent.trim().replace(/\n+/g, br).replace(/\|/g, '\\|');
|
|
389
764
|
rowCells.push(cellContent);
|
|
390
765
|
// Handle colspan by adding empty cells
|
|
@@ -460,7 +835,11 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
460
835
|
rows += await this.renderTableAsHtml(row, await this.handleOnNode(row));
|
|
461
836
|
}
|
|
462
837
|
}
|
|
463
|
-
|
|
838
|
+
// Carry table-layout alignment through the HTML fallback so it isn't lost
|
|
839
|
+
// just because the table also needed HTML for merged cells.
|
|
840
|
+
const tableMeta = node.metadata;
|
|
841
|
+
const alignAttr = tableMeta?.align ? ` data-align="${(0, sanitize_js_1.escapeHtml)(tableMeta.align)}"` : '';
|
|
842
|
+
return `<table${alignAttr}>\n${rows}</table>\n`;
|
|
464
843
|
}
|
|
465
844
|
else if (node.type === 'row') {
|
|
466
845
|
let cells = '';
|
|
@@ -482,7 +861,9 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
482
861
|
content += await this.processNodeRecursive(child, async (n, co) => {
|
|
483
862
|
switch (n.type) {
|
|
484
863
|
case 'text': {
|
|
485
|
-
|
|
864
|
+
// Inside HTML table cells, entity-encode angle brackets so cell
|
|
865
|
+
// text can't inject a raw tag (e.g. </td><script>).
|
|
866
|
+
let text = (0, sanitize_js_1.markdownEscapeText)(n.text || '');
|
|
486
867
|
if (n.formatting?.bold)
|
|
487
868
|
text = `<b>${text}</b>`;
|
|
488
869
|
if (n.formatting?.italic)
|
|
@@ -497,7 +878,7 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
|
|
|
497
878
|
}
|
|
498
879
|
case 'paragraph': return `<p>${co}</p>`;
|
|
499
880
|
case 'heading': {
|
|
500
|
-
const level = n.metadata?.level || 1;
|
|
881
|
+
const level = Math.min(Math.max(Number(n.metadata?.level) || 1, 1), 6);
|
|
501
882
|
return `<h${level}>${co}</h${level}>`;
|
|
502
883
|
}
|
|
503
884
|
case 'table': return await this.renderTableAsHtml(n);
|