@brett_lamy/docstream 1.2.3 → 1.2.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +5 -1
- package/package.json +1 -1
- package/src/demo/markdown.ts +1 -1
- package/src/docs/DocsRenderer.tsx +21 -9
- package/src/gitbook/ast.ts +240 -43
- package/src/gitbook/inline.ts +542 -149
- package/src/gitbook/parse.ts +403 -165
- package/src/gitbook/serialize.ts +446 -126
package/src/gitbook/inline.ts
CHANGED
|
@@ -1,15 +1,20 @@
|
|
|
1
|
-
import type { Inline, InlineDelimiter, InlineImageNode, ReferenceNode, TextNode } from "./ast"
|
|
1
|
+
import type { Inline, InlineDelimiter, InlineImageNode, LinkForm, ReferenceNode, TextNode } from "./ast"
|
|
2
2
|
|
|
3
3
|
type Marks = Partial<Omit<TextNode, "type" | "text">>
|
|
4
4
|
|
|
5
5
|
// Reference-style link definitions ([ref]: url), populated by parseMarkdown
|
|
6
|
-
// and consumed here for [text][ref] / [text][] forms.
|
|
6
|
+
// and consumed here for [text][ref] / [text][] / [text] forms. Keys are normalized labels.
|
|
7
7
|
export const refDefinitions = new Map<string, string>()
|
|
8
8
|
|
|
9
9
|
// Footnote citation definitions ([^id]: url "Label"), populated by
|
|
10
10
|
// parseMarkdown and consumed here to resolve [^id] markers.
|
|
11
11
|
export const footnoteDefinitions = new Map<string, { url: string; label?: string }>()
|
|
12
12
|
|
|
13
|
+
/** A reference label as matched against definitions: case-insensitive, inner whitespace collapsed. */
|
|
14
|
+
export function normalizeLabel(label: string): string {
|
|
15
|
+
return label.trim().replace(/\s+/g, " ").toLowerCase()
|
|
16
|
+
}
|
|
17
|
+
|
|
13
18
|
// @mention / #tag / $codebase body: letter/underscore start, then word chars, dots,
|
|
14
19
|
// dashes. Codebase ids may also carry `/` (`$org/repo`). A letter start keeps prices
|
|
15
20
|
// like $5 plain text.
|
|
@@ -22,9 +27,11 @@ const KIND_SIGIL = { mention: "@", tag: "#", codebase: "$" } as const
|
|
|
22
27
|
const isReferenceBoundary = (prev: string | undefined) =>
|
|
23
28
|
prev === undefined || /[\s([{*_~]/.test(prev)
|
|
24
29
|
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
30
|
+
type ImgAttrs = Pick<InlineImageNode, "src" | "alt" | "width" | "height">
|
|
31
|
+
|
|
32
|
+
function imgAttrs(attrStr: string): ImgAttrs {
|
|
33
|
+
const attr = (name: string) => attrStr.match(new RegExp(`(?:^|\\s)${name}="([^"]*)"`, "i"))?.[1]
|
|
34
|
+
const out: ImgAttrs = { src: attr("src") ?? "" }
|
|
28
35
|
const alt = attr("alt")
|
|
29
36
|
const width = attr("width")
|
|
30
37
|
const height = attr("height")
|
|
@@ -34,19 +41,42 @@ function imgAttrs(attrStr: string): Omit<InlineImageNode, "type" | "link"> {
|
|
|
34
41
|
return out
|
|
35
42
|
}
|
|
36
43
|
|
|
44
|
+
const LINKED_IMG_RE = /^<a\s[^>]*href="([^"]*)"[^>]*>\s*<img\s([^>]*?)\/?>\s*<\/a>/i
|
|
45
|
+
const HTML_IMG_RE = /^<img\s([^>]*?)\/?>/i
|
|
46
|
+
|
|
47
|
+
/** What an image's recorded HTML says — `null` when it isn't an (optionally linked) `<img>`. */
|
|
48
|
+
function htmlImage(html: string): (ImgAttrs & { link?: string }) | null {
|
|
49
|
+
const linked = html.match(LINKED_IMG_RE)
|
|
50
|
+
if (linked && linked[0] === html) return { ...imgAttrs(linked[2]), link: linked[1] }
|
|
51
|
+
const bare = html.match(HTML_IMG_RE)
|
|
52
|
+
if (bare && bare[0] === html) return imgAttrs(bare[1])
|
|
53
|
+
return null
|
|
54
|
+
}
|
|
55
|
+
|
|
37
56
|
// ─── Parsing ─────────────────────────────────────────────────────────────────────────────────────────────
|
|
38
57
|
// Two passes, after CommonMark: tokenize (escapes, code spans, links, images, chips, autolinks and `*` / `_` /
|
|
39
58
|
// `~~` delimiter runs with their flanking), then pair the delimiter runs with the CommonMark "process
|
|
40
|
-
// emphasis" algorithm (rule of 3 included), which yields a tree. The tree is flattened into
|
|
41
|
-
// remembers how its formatting was spelled (`delims
|
|
59
|
+
// emphasis" algorithm (rule of 3 included), which yields a tree. The tree is flattened into inlines that all
|
|
60
|
+
// carry their marks; each remembers how its formatting was spelled (`delims`, link form, title, reference
|
|
61
|
+
// label, HTML tag) whenever that differs from the default output, and escaped characters stay their own runs.
|
|
42
62
|
|
|
43
63
|
type Mark = "bold" | "italic" | "strike" | "link"
|
|
44
64
|
|
|
65
|
+
/** A link as written: target plus the form-specific details. */
|
|
66
|
+
interface LinkInfo {
|
|
67
|
+
url: string
|
|
68
|
+
title?: string
|
|
69
|
+
ref?: string
|
|
70
|
+
tag?: string
|
|
71
|
+
close?: string
|
|
72
|
+
}
|
|
73
|
+
|
|
45
74
|
type Tree =
|
|
46
75
|
| { k: "text"; text: string }
|
|
47
|
-
| { k: "
|
|
76
|
+
| { k: "esc"; text: string }
|
|
77
|
+
| { k: "code"; text: string; ticks?: number }
|
|
48
78
|
| { k: "atom"; node: InlineImageNode | ReferenceNode }
|
|
49
|
-
| { k: "wrap"; delim: InlineDelimiter;
|
|
79
|
+
| { k: "wrap"; delim: InlineDelimiter; link?: LinkInfo; children: Tree[] }
|
|
50
80
|
|
|
51
81
|
interface DelimRun {
|
|
52
82
|
k: "delim"
|
|
@@ -70,8 +100,16 @@ const MARK_OF: Record<InlineDelimiter, Mark> = {
|
|
|
70
100
|
link: "link",
|
|
71
101
|
"<>": "link",
|
|
72
102
|
url: "link",
|
|
103
|
+
ref: "link",
|
|
104
|
+
collapsed: "link",
|
|
105
|
+
shortcut: "link",
|
|
106
|
+
html: "link",
|
|
73
107
|
}
|
|
74
108
|
|
|
109
|
+
const isLinkForm = (d: InlineDelimiter): d is LinkForm => MARK_OF[d] === "link"
|
|
110
|
+
/** Link forms whose text sits between `[` and `]`. */
|
|
111
|
+
const BRACKETED = new Set<InlineDelimiter>(["link", "ref", "collapsed", "shortcut"])
|
|
112
|
+
|
|
75
113
|
const ASCII_PUNCT = /[!-/:-@[-`{-~]/
|
|
76
114
|
const isSpace = (c: string | undefined) => c === undefined || /\s/.test(c)
|
|
77
115
|
const isPunct = (c: string | undefined) => c !== undefined && /[\p{P}\p{S}]/u.test(c)
|
|
@@ -106,23 +144,53 @@ function closingBackticks(src: string, from: number, n: number): number {
|
|
|
106
144
|
return -1
|
|
107
145
|
}
|
|
108
146
|
|
|
109
|
-
|
|
110
|
-
|
|
147
|
+
/** Index of the `]` closing the `[` at `start` — brackets nest, escapes and code spans are skipped — or -1. */
|
|
148
|
+
function closingBracket(src: string, start: number): number {
|
|
149
|
+
let depth = 0
|
|
150
|
+
for (let j = start; j < src.length; j++) {
|
|
151
|
+
const c = src[j]
|
|
152
|
+
if (c === "\\") {
|
|
153
|
+
j++
|
|
154
|
+
continue
|
|
155
|
+
}
|
|
156
|
+
if (c === "`") {
|
|
157
|
+
let n = 1
|
|
158
|
+
while (src[j + n] === "`") n++
|
|
159
|
+
const end = closingBackticks(src, j + n, n)
|
|
160
|
+
j = (end === -1 ? j + n : end + n) - 1
|
|
161
|
+
continue
|
|
162
|
+
}
|
|
163
|
+
if (c === "[") depth++
|
|
164
|
+
else if (c === "]" && --depth === 0) return j
|
|
165
|
+
}
|
|
166
|
+
return -1
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
/** `(url)` or `(url "title")` at the start of `rest`. */
|
|
170
|
+
const LINK_TAIL_RE = /^\(([^)\s]+)(?:[ \t]+"((?:\\.|[^"\\])*)")?\)/
|
|
171
|
+
const HTML_LINK_RE = /^(<a\s[^>]*?href="([^"]*)"[^>]*>)(.*?)(<\/a\s*>)/i
|
|
172
|
+
|
|
173
|
+
interface Ctx {
|
|
174
|
+
defs: ReadonlyMap<string, string>
|
|
175
|
+
}
|
|
111
176
|
|
|
112
177
|
/** `inLink`: tokenizing link text, where CommonMark allows no further links (or autolinks). */
|
|
113
|
-
function tokenize(src: string, inLink = false): Tok[] {
|
|
178
|
+
function tokenize(src: string, ctx: Ctx, inLink = false): Tok[] {
|
|
114
179
|
const toks: Tok[] = []
|
|
115
180
|
let buf = ""
|
|
116
|
-
const
|
|
181
|
+
const flush = () => {
|
|
117
182
|
if (buf) toks.push({ k: "text", text: buf })
|
|
118
183
|
buf = ""
|
|
184
|
+
}
|
|
185
|
+
const push = (t: Tok) => {
|
|
186
|
+
flush()
|
|
119
187
|
toks.push(t)
|
|
120
188
|
}
|
|
121
|
-
const wrap = (delim: InlineDelimiter,
|
|
189
|
+
const wrap = (delim: InlineDelimiter, link: LinkInfo, inner: string | Tree[]): Tok => ({
|
|
122
190
|
k: "wrap",
|
|
123
191
|
delim,
|
|
124
|
-
|
|
125
|
-
children: typeof inner === "string" ? pairEmphasis(tokenize(inner, true)) : inner,
|
|
192
|
+
link,
|
|
193
|
+
children: typeof inner === "string" ? pairEmphasis(tokenize(inner, ctx, true)) : inner,
|
|
126
194
|
})
|
|
127
195
|
|
|
128
196
|
let i = 0
|
|
@@ -130,9 +198,13 @@ function tokenize(src: string, inLink = false): Tok[] {
|
|
|
130
198
|
const c = src[i]
|
|
131
199
|
const rest = src.slice(i)
|
|
132
200
|
|
|
133
|
-
// Backslash escapes: ASCII punctuation only; `C:\path` keeps its backslash.
|
|
201
|
+
// Backslash escapes: ASCII punctuation only; `C:\path` keeps its backslash. Escaped characters are their
|
|
202
|
+
// own run, so the author's backslashes come back even where they aren't needed.
|
|
134
203
|
if (c === "\\" && i + 1 < src.length && ASCII_PUNCT.test(src[i + 1])) {
|
|
135
|
-
|
|
204
|
+
flush()
|
|
205
|
+
const last = toks[toks.length - 1]
|
|
206
|
+
if (last?.k === "esc") last.text += src[i + 1]
|
|
207
|
+
else toks.push({ k: "esc", text: src[i + 1] })
|
|
136
208
|
i += 2
|
|
137
209
|
continue
|
|
138
210
|
}
|
|
@@ -143,7 +215,7 @@ function tokenize(src: string, inLink = false): Tok[] {
|
|
|
143
215
|
while (src[i + n] === "`") n++
|
|
144
216
|
const end = closingBackticks(src, i + n, n)
|
|
145
217
|
if (end !== -1) {
|
|
146
|
-
push({ k: "code", text: src.slice(i + n, end) })
|
|
218
|
+
push({ k: "code", text: src.slice(i + n, end), ...(n > 1 ? { ticks: n } : {}) })
|
|
147
219
|
i = end + n
|
|
148
220
|
} else {
|
|
149
221
|
buf += src.slice(i, i + n)
|
|
@@ -154,41 +226,66 @@ function tokenize(src: string, inLink = false): Tok[] {
|
|
|
154
226
|
|
|
155
227
|
if (c === "<") {
|
|
156
228
|
// <a href="…"><img …></a> — linked image (GitHub badge style)
|
|
157
|
-
const linkedImg = rest.match(
|
|
229
|
+
const linkedImg = !inLink && rest.match(LINKED_IMG_RE)
|
|
158
230
|
if (linkedImg) {
|
|
159
|
-
push({ k: "atom", node: { type: "image", ...imgAttrs(linkedImg[2]), link: linkedImg[1] } })
|
|
231
|
+
push({ k: "atom", node: { type: "image", ...imgAttrs(linkedImg[2]), link: linkedImg[1], html: linkedImg[0] } })
|
|
160
232
|
i += linkedImg[0].length
|
|
161
233
|
continue
|
|
162
234
|
}
|
|
163
235
|
// <img …> — bare inline image
|
|
164
|
-
const htmlImg = rest.match(
|
|
236
|
+
const htmlImg = rest.match(HTML_IMG_RE)
|
|
165
237
|
if (htmlImg) {
|
|
166
|
-
push({ k: "atom", node: { type: "image", ...imgAttrs(htmlImg[1]) } })
|
|
238
|
+
push({ k: "atom", node: { type: "image", ...imgAttrs(htmlImg[1]), html: htmlImg[0] } })
|
|
167
239
|
i += htmlImg[0].length
|
|
168
240
|
continue
|
|
169
241
|
}
|
|
170
|
-
// <a href="…">text</a> — html link
|
|
171
|
-
const htmlLink = !inLink && rest.match(
|
|
242
|
+
// <a href="…">text</a> — html link, its opening tag kept as written
|
|
243
|
+
const htmlLink = !inLink && rest.match(HTML_LINK_RE)
|
|
172
244
|
if (htmlLink) {
|
|
173
|
-
push(wrap("
|
|
245
|
+
push(wrap("html", { url: htmlLink[2], tag: htmlLink[1], ...(htmlLink[4] !== "</a>" ? { close: htmlLink[4] } : {}) }, htmlLink[3]))
|
|
174
246
|
i += htmlLink[0].length
|
|
175
247
|
continue
|
|
176
248
|
}
|
|
177
249
|
// <https://…> — angle-bracket autolink
|
|
178
250
|
const angleLink = !inLink && rest.match(/^<(https?:\/\/[^>\s]+)>/i)
|
|
179
251
|
if (angleLink) {
|
|
180
|
-
push(wrap("<>", angleLink[1], [{ k: "text", text: angleLink[1] }]))
|
|
252
|
+
push(wrap("<>", { url: angleLink[1] }, [{ k: "text", text: angleLink[1] }]))
|
|
181
253
|
i += angleLink[0].length
|
|
182
254
|
continue
|
|
183
255
|
}
|
|
184
256
|
}
|
|
185
257
|
|
|
186
|
-
//  — markdown inline image
|
|
187
|
-
if (c === "!") {
|
|
188
|
-
const
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
258
|
+
//  — markdown inline image; ![alt][ref] / ![alt][] / ![alt] through a definition
|
|
259
|
+
if (c === "!" && src[i + 1] === "[") {
|
|
260
|
+
const close = closingBracket(src, i + 1)
|
|
261
|
+
const tail = close !== -1 ? src.slice(close + 1).match(LINK_TAIL_RE) : null
|
|
262
|
+
const alt = close !== -1 ? src.slice(i + 2, close) : ""
|
|
263
|
+
if (close !== -1 && !tail) {
|
|
264
|
+
const ref = src.slice(close + 1).match(/^\[((?:\\.|[^\]\\])*)\]/)
|
|
265
|
+
const label = ref ? ref[1] || alt : alt
|
|
266
|
+
const url = label.trim() ? ctx.defs.get(normalizeLabel(label)) : undefined
|
|
267
|
+
if (url !== undefined) {
|
|
268
|
+
const srcForm = ref ? (ref[1] ? "ref" : "collapsed") : "shortcut"
|
|
269
|
+
push({
|
|
270
|
+
k: "atom",
|
|
271
|
+
node: { type: "image", src: url, ...(alt ? { alt } : {}), syntax: "markdown", srcForm, srcRef: label },
|
|
272
|
+
})
|
|
273
|
+
i = close + 1 + (ref ? ref[0].length : 0)
|
|
274
|
+
continue
|
|
275
|
+
}
|
|
276
|
+
}
|
|
277
|
+
if (tail) {
|
|
278
|
+
push({
|
|
279
|
+
k: "atom",
|
|
280
|
+
node: {
|
|
281
|
+
type: "image",
|
|
282
|
+
src: tail[1],
|
|
283
|
+
...(alt ? { alt } : {}),
|
|
284
|
+
...(tail[2] !== undefined ? { title: tail[2] } : {}),
|
|
285
|
+
syntax: "markdown",
|
|
286
|
+
},
|
|
287
|
+
})
|
|
288
|
+
i = close + 1 + tail[0].length
|
|
192
289
|
continue
|
|
193
290
|
}
|
|
194
291
|
}
|
|
@@ -202,20 +299,30 @@ function tokenize(src: string, inLink = false): Tok[] {
|
|
|
202
299
|
i += cite[0].length
|
|
203
300
|
continue
|
|
204
301
|
}
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
302
|
+
const close = inLink ? -1 : closingBracket(src, i)
|
|
303
|
+
if (close !== -1) {
|
|
304
|
+
const text = src.slice(i + 1, close)
|
|
305
|
+
const after = src.slice(close + 1)
|
|
306
|
+
// [text](url "title") — link text is its own emphasis scope, as in CommonMark
|
|
307
|
+
const tail = after.match(LINK_TAIL_RE)
|
|
308
|
+
if (tail) {
|
|
309
|
+
push(wrap("link", { url: tail[1], ...(tail[2] !== undefined ? { title: tail[2] } : {}) }, text))
|
|
310
|
+
i = close + 1 + tail[0].length
|
|
311
|
+
continue
|
|
312
|
+
}
|
|
313
|
+
// [text][ref] / [text][] — reference-style links
|
|
314
|
+
const ref = after.match(/^\[((?:\\.|[^\]\\])*)\]/)
|
|
315
|
+
const refUrl = ref ? ctx.defs.get(normalizeLabel(ref[1] || text)) : undefined
|
|
316
|
+
if (ref && refUrl !== undefined) {
|
|
317
|
+
push(wrap(ref[1] ? "ref" : "collapsed", { url: refUrl, ref: ref[1] || text }, text))
|
|
318
|
+
i = close + 1 + ref[0].length
|
|
319
|
+
continue
|
|
320
|
+
}
|
|
321
|
+
// [ref] — shortcut reference link
|
|
322
|
+
const shortcut = text.trim() ? ctx.defs.get(normalizeLabel(text)) : undefined
|
|
323
|
+
if (shortcut !== undefined) {
|
|
324
|
+
push(wrap("shortcut", { url: shortcut, ref: text }, text))
|
|
325
|
+
i = close + 1
|
|
219
326
|
continue
|
|
220
327
|
}
|
|
221
328
|
}
|
|
@@ -238,7 +345,7 @@ function tokenize(src: string, inLink = false): Tok[] {
|
|
|
238
345
|
if (!inLink && (c === "h" || c === "H") && /^https?:\/\//i.test(rest) && (i === 0 || /[\s*_~]/.test(src[i - 1]))) {
|
|
239
346
|
const m = rest.match(/^https?:\/\/[^\s<>"')\]]+/i)!
|
|
240
347
|
const url = m[0].replace(/[.,;:!?*_~]+$/, "")
|
|
241
|
-
push(wrap("url", url, [{ k: "text", text: url }]))
|
|
348
|
+
push(wrap("url", { url }, [{ k: "text", text: url }]))
|
|
242
349
|
i += url.length
|
|
243
350
|
continue
|
|
244
351
|
}
|
|
@@ -260,7 +367,7 @@ function tokenize(src: string, inLink = false): Tok[] {
|
|
|
260
367
|
buf += c
|
|
261
368
|
i++
|
|
262
369
|
}
|
|
263
|
-
|
|
370
|
+
flush()
|
|
264
371
|
return toks
|
|
265
372
|
}
|
|
266
373
|
|
|
@@ -316,67 +423,186 @@ function pairEmphasis(toks: Tok[]): Tree[] {
|
|
|
316
423
|
return settle(toks)
|
|
317
424
|
}
|
|
318
425
|
|
|
319
|
-
/**
|
|
320
|
-
function
|
|
426
|
+
/** Is the inline's `link` a link mark (rather than the HTML anchor of an `<a href><img></a>` image)? */
|
|
427
|
+
function linkIsMark(n: Inline): boolean {
|
|
428
|
+
if (!n.link) return false
|
|
429
|
+
return n.type !== "image" || !!n.delims?.some(isLinkForm)
|
|
430
|
+
}
|
|
431
|
+
|
|
432
|
+
/** The spelling an inline gets when nothing says otherwise — what serialization produced before `delims`. */
|
|
433
|
+
function defaultDelims(n: Inline): InlineDelimiter[] {
|
|
321
434
|
const emphasis: InlineDelimiter[] = []
|
|
322
435
|
if (n.strike) emphasis.push("~~")
|
|
323
436
|
if (n.italic) emphasis.push("_")
|
|
324
437
|
if (n.bold) emphasis.push("**")
|
|
325
|
-
|
|
438
|
+
// An image's link is its HTML anchor unless the spelling says otherwise.
|
|
439
|
+
if (!n.link || n.type === "image") return emphasis
|
|
326
440
|
const inner = !!n.linkInner && emphasis.length > 0
|
|
327
|
-
const form: InlineDelimiter =
|
|
441
|
+
const form: InlineDelimiter =
|
|
442
|
+
n.type === "text" && (inner || !emphasis.length) && n.text === n.link && !n.code && !n.escaped ? "url" : "link"
|
|
328
443
|
return inner ? [...emphasis, form] : [form, ...emphasis]
|
|
329
444
|
}
|
|
330
445
|
|
|
331
446
|
const sameList = (a: readonly string[], b: readonly string[]) => a.length === b.length && a.every((x, k) => x === b[k])
|
|
332
447
|
|
|
333
|
-
|
|
448
|
+
interface PathEntry {
|
|
449
|
+
delim: InlineDelimiter
|
|
450
|
+
link?: LinkInfo
|
|
451
|
+
/** Which link (wrap) this is, for link identity. */
|
|
452
|
+
seq?: number
|
|
453
|
+
}
|
|
454
|
+
|
|
455
|
+
interface FlattenState {
|
|
456
|
+
seq: number
|
|
457
|
+
/** The link wrap each output inline belongs to. */
|
|
458
|
+
linkSeq: Map<Inline, number>
|
|
459
|
+
}
|
|
460
|
+
|
|
461
|
+
function applyPath(n: Inline, path: PathEntry[], state: FlattenState) {
|
|
462
|
+
let emphasis = false
|
|
463
|
+
for (const e of path) {
|
|
464
|
+
const mark = MARK_OF[e.delim]
|
|
465
|
+
if (mark === "link") {
|
|
466
|
+
const link = e.link!
|
|
467
|
+
n.link = link.url
|
|
468
|
+
if (link.title !== undefined) n.linkTitle = link.title
|
|
469
|
+
if (link.ref !== undefined) n.linkRef = link.ref
|
|
470
|
+
if (link.tag !== undefined) n.linkTag = link.tag
|
|
471
|
+
if (link.close !== undefined) n.linkTagEnd = link.close
|
|
472
|
+
if (emphasis) n.linkInner = true
|
|
473
|
+
state.linkSeq.set(n, e.seq!)
|
|
474
|
+
} else {
|
|
475
|
+
n[mark] = true
|
|
476
|
+
emphasis = true
|
|
477
|
+
}
|
|
478
|
+
}
|
|
479
|
+
const delims = path.map((e) => e.delim)
|
|
480
|
+
if (!sameList(delims, defaultDelims(n))) n.delims = delims
|
|
481
|
+
}
|
|
482
|
+
|
|
483
|
+
function flatten(trees: Tree[], path: PathEntry[], base: Marks, out: Inline[], state: FlattenState) {
|
|
334
484
|
for (const t of trees) {
|
|
335
485
|
if (t.k === "wrap") {
|
|
336
|
-
|
|
486
|
+
const entry: PathEntry = { delim: t.delim, ...(t.link ? { link: t.link, seq: ++state.seq } : {}) }
|
|
487
|
+
const before = out.length
|
|
488
|
+
flatten(t.children, [...path, entry], base, out, state)
|
|
489
|
+
// An empty link (`[](url)`) is an empty run, so it isn't lost.
|
|
490
|
+
if (t.link && out.length === before) {
|
|
491
|
+
const n: TextNode = { type: "text", text: "", ...base }
|
|
492
|
+
applyPath(n, [...path, entry], state)
|
|
493
|
+
out.push(n)
|
|
494
|
+
}
|
|
337
495
|
continue
|
|
338
496
|
}
|
|
339
497
|
if (t.k === "atom") {
|
|
340
|
-
|
|
498
|
+
const node = { ...t.node }
|
|
499
|
+
applyPath(node, path, state)
|
|
500
|
+
out.push(node)
|
|
341
501
|
continue
|
|
342
502
|
}
|
|
343
503
|
const n: TextNode = { type: "text", text: t.text, ...base }
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
if (mark === "link") {
|
|
348
|
-
n.link = e.url
|
|
349
|
-
if (emphasis) n.linkInner = true
|
|
350
|
-
} else {
|
|
351
|
-
n[mark] = true
|
|
352
|
-
emphasis = true
|
|
353
|
-
}
|
|
504
|
+
if (t.k === "code") {
|
|
505
|
+
n.code = true
|
|
506
|
+
if (t.ticks && t.ticks !== codeSpanTicks(t.text)) n.ticks = t.ticks
|
|
354
507
|
}
|
|
355
|
-
if (t.k === "
|
|
356
|
-
|
|
357
|
-
if (!sameList(delims, defaultDelims(n))) n.delims = delims
|
|
508
|
+
if (t.k === "esc") n.escaped = true
|
|
509
|
+
applyPath(n, path, state)
|
|
358
510
|
out.push(n)
|
|
359
511
|
}
|
|
360
512
|
}
|
|
361
513
|
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
514
|
+
/** What makes two link runs the same link to the serializer (everything but identity). */
|
|
515
|
+
function linkKey(n: Inline): string {
|
|
516
|
+
const form = inlineDelims(n).find(isLinkForm) ?? ""
|
|
517
|
+
return JSON.stringify([n.link, form, n.linkTitle ?? null, n.linkRef ?? null, n.linkTag ?? null, n.linkTagEnd ?? null])
|
|
518
|
+
}
|
|
519
|
+
|
|
520
|
+
/** Adjacent links to the same target stay distinct: the later one gets a `linkId`. */
|
|
521
|
+
function markLinkIdentity(out: Inline[], state: FlattenState) {
|
|
522
|
+
let next = 1
|
|
523
|
+
const ids = new Map<number, number>()
|
|
524
|
+
for (let k = 1; k < out.length; k++) {
|
|
525
|
+
const a = out[k - 1]
|
|
526
|
+
const b = out[k]
|
|
527
|
+
const sa = state.linkSeq.get(a)
|
|
528
|
+
const sb = state.linkSeq.get(b)
|
|
529
|
+
if (sa === undefined || sb === undefined || sa === sb || !linkIsMark(a) || !linkIsMark(b)) continue
|
|
530
|
+
if (a.linkId !== undefined || linkKey(a) !== linkKey(b)) continue
|
|
531
|
+
if (!ids.has(sb)) ids.set(sb, next++)
|
|
532
|
+
}
|
|
533
|
+
if (!ids.size) return
|
|
534
|
+
for (const n of out) {
|
|
535
|
+
const id = ids.get(state.linkSeq.get(n) ?? -1)
|
|
536
|
+
if (id !== undefined) n.linkId = id
|
|
537
|
+
}
|
|
538
|
+
}
|
|
539
|
+
|
|
540
|
+
function parseWith(src: string, ctx: Ctx, marks: Marks = {}): Inline[] {
|
|
367
541
|
const out: Inline[] = []
|
|
368
|
-
|
|
542
|
+
const state: FlattenState = { seq: 0, linkSeq: new Map() }
|
|
543
|
+
flatten(pairEmphasis(tokenize(src, ctx)), [], marks, out, state)
|
|
544
|
+
markLinkIdentity(out, state)
|
|
369
545
|
return out
|
|
370
546
|
}
|
|
371
547
|
|
|
548
|
+
// Parses GitBook/GFM inline markdown into flat inlines with marks.
|
|
549
|
+
// Supported: **bold** / __bold__, _italic_ / *italic*, ***both***, ~~strike~~, `code`, [text](url "title"),
|
|
550
|
+
// [text][ref] / [text][] / [ref], <url> and bare URLs, @mention / #tag / $codebase / [^n] chips,
|
|
551
|
+
// , backslash escapes, plus GitHub-style inline HTML: <img …>, <a href><img …></a>,
|
|
552
|
+
// <a href>text</a>.
|
|
553
|
+
export function parseInline(src: string, marks: Marks = {}): Inline[] {
|
|
554
|
+
return parseWith(src, { defs: refDefinitions }, marks)
|
|
555
|
+
}
|
|
556
|
+
|
|
372
557
|
// ─── Serializing ─────────────────────────────────────────────────────────────────────────────────────────
|
|
373
558
|
|
|
559
|
+
/**
|
|
560
|
+
* Definitions of the document being serialized (normalized label → url), so reference-style links whose
|
|
561
|
+
* target no longer matches their definition fall back to inline links. Set by `serializeMarkdown`.
|
|
562
|
+
*/
|
|
563
|
+
let documentDefs: ReadonlyMap<string, string> | null = null
|
|
564
|
+
|
|
565
|
+
/** Run `fn` with the document's reference definitions in scope for `serializeInline`. */
|
|
566
|
+
export function withDocumentDefinitions<T>(defs: ReadonlyMap<string, string>, fn: () => T): T {
|
|
567
|
+
const saved = documentDefs
|
|
568
|
+
documentDefs = defs
|
|
569
|
+
try {
|
|
570
|
+
return fn()
|
|
571
|
+
} finally {
|
|
572
|
+
documentDefs = saved
|
|
573
|
+
}
|
|
574
|
+
}
|
|
575
|
+
|
|
374
576
|
export function serializeReference(n: ReferenceNode): string {
|
|
375
577
|
if (n.kind === "citation") return `[^${n.id}]`
|
|
376
578
|
return `${KIND_SIGIL[n.kind]}${n.id}`
|
|
377
579
|
}
|
|
378
580
|
|
|
379
|
-
function serializeImage(n: InlineImageNode): string {
|
|
581
|
+
function serializeImage(n: InlineImageNode, atomicLink: boolean): string {
|
|
582
|
+
const link = atomicLink ? n.link : undefined
|
|
583
|
+
const alt = n.alt ?? ""
|
|
584
|
+
if (n.syntax === "markdown" && !link && !alt.includes("\n") && closingBracket(`[${alt}]`, 0) === alt.length + 1) {
|
|
585
|
+
// Through its definition, while that still gives this source.
|
|
586
|
+
if (n.srcForm && n.srcRef !== undefined && (!documentDefs || documentDefs.get(normalizeLabel(n.srcRef)) === n.src)) {
|
|
587
|
+
if (n.srcForm === "ref") return `![${alt}][${n.srcRef}]`
|
|
588
|
+
const same = normalizeLabel(unescapeLabel(n.srcRef)) === normalizeLabel(unescapeLabel(alt))
|
|
589
|
+
return !same ? `![${alt}][${n.srcRef}]` : n.srcForm === "collapsed" ? `![${alt}][]` : `![${alt}]`
|
|
590
|
+
}
|
|
591
|
+
return ``
|
|
592
|
+
}
|
|
593
|
+
if (n.html) {
|
|
594
|
+
const said = htmlImage(n.html)
|
|
595
|
+
if (
|
|
596
|
+
said &&
|
|
597
|
+
said.src === n.src &&
|
|
598
|
+
(said.alt ?? "") === (n.alt ?? "") &&
|
|
599
|
+
(said.width ?? "") === (n.width ?? "") &&
|
|
600
|
+
(said.height ?? "") === (n.height ?? "") &&
|
|
601
|
+
(said.link ?? "") === (link ?? "")
|
|
602
|
+
) {
|
|
603
|
+
return n.html
|
|
604
|
+
}
|
|
605
|
+
}
|
|
380
606
|
const attrs = [
|
|
381
607
|
`src="${n.src}"`,
|
|
382
608
|
n.alt ? `alt="${n.alt}"` : "",
|
|
@@ -386,25 +612,30 @@ function serializeImage(n: InlineImageNode): string {
|
|
|
386
612
|
.filter(Boolean)
|
|
387
613
|
.join(" ")
|
|
388
614
|
const img = `<img ${attrs} />`
|
|
389
|
-
return
|
|
615
|
+
return link ? `<a href="${link}">${img}</a>` : img
|
|
390
616
|
}
|
|
391
617
|
|
|
392
618
|
interface Entry {
|
|
393
619
|
mark: Mark
|
|
394
620
|
delim: InlineDelimiter
|
|
395
621
|
url?: string
|
|
622
|
+
title?: string
|
|
623
|
+
ref?: string
|
|
624
|
+
tag?: string
|
|
625
|
+
tagEnd?: string
|
|
626
|
+
id?: number
|
|
396
627
|
}
|
|
397
628
|
|
|
398
|
-
/** The
|
|
399
|
-
function entriesFor(n:
|
|
629
|
+
/** The inline's formatting, outermost → innermost: its `delims` hint reconciled with its actual marks. */
|
|
630
|
+
function entriesFor(n: Inline): Entry[] {
|
|
400
631
|
const fallback = defaultDelims(n)
|
|
401
632
|
let delims = fallback
|
|
402
633
|
if (n.delims?.length) {
|
|
403
|
-
const has = (m: Mark) => (m === "link" ?
|
|
634
|
+
const has = (m: Mark) => (m === "link" ? linkIsMark(n) : !!n[m])
|
|
404
635
|
delims = []
|
|
405
636
|
for (const d of n.delims) {
|
|
406
637
|
const m = MARK_OF[d]
|
|
407
|
-
if (m && has(m) && !(m === "link" && delims.some(
|
|
638
|
+
if (m && has(m) && !(m === "link" && delims.some(isLinkForm))) delims.push(d)
|
|
408
639
|
}
|
|
409
640
|
// Marks without a spelling: emphasis innermost (inside a trailing link), the link where it defaults to.
|
|
410
641
|
for (const d of fallback) {
|
|
@@ -415,38 +646,116 @@ function entriesFor(n: TextNode): Entry[] {
|
|
|
415
646
|
else delims.unshift(d)
|
|
416
647
|
} else {
|
|
417
648
|
const last = delims[delims.length - 1]
|
|
418
|
-
delims.splice(last &&
|
|
649
|
+
delims.splice(last && isLinkForm(last) ? delims.length - 1 : delims.length, 0, d)
|
|
419
650
|
}
|
|
420
651
|
}
|
|
652
|
+
delims = delims.map((d, k) => validForm(n, d, k === delims.length - 1))
|
|
653
|
+
}
|
|
654
|
+
return delims.map((delim) => {
|
|
655
|
+
if (!isLinkForm(delim)) return { mark: MARK_OF[delim], delim }
|
|
656
|
+
const e: Entry = { mark: "link", delim, url: n.link }
|
|
657
|
+
if (delim === "link" && n.linkTitle !== undefined) e.title = n.linkTitle
|
|
658
|
+
if ((delim === "ref" || delim === "collapsed" || delim === "shortcut") && n.linkRef !== undefined) e.ref = n.linkRef
|
|
659
|
+
if (delim === "html") {
|
|
660
|
+
e.tag = n.linkTag
|
|
661
|
+
if (n.linkTagEnd && /^<\/a\s*>$/i.test(n.linkTagEnd)) e.tagEnd = n.linkTagEnd
|
|
662
|
+
}
|
|
663
|
+
if (n.linkId !== undefined) e.id = n.linkId
|
|
664
|
+
return e
|
|
665
|
+
})
|
|
666
|
+
}
|
|
667
|
+
|
|
668
|
+
/** A link form the inline can still be written in, or `"link"`. */
|
|
669
|
+
function validForm(n: Inline, d: InlineDelimiter, innermost: boolean): InlineDelimiter {
|
|
670
|
+
switch (d) {
|
|
421
671
|
// An autolink shows its URL as its text: only valid innermost, around an unchanged URL.
|
|
422
|
-
|
|
423
|
-
|
|
424
|
-
|
|
672
|
+
case "url":
|
|
673
|
+
case "<>":
|
|
674
|
+
return innermost && n.type === "text" && n.text === n.link && !n.code && !n.escaped ? d : "link"
|
|
675
|
+
case "ref":
|
|
676
|
+
case "collapsed":
|
|
677
|
+
case "shortcut": {
|
|
678
|
+
if (n.linkRef === undefined || (d === "ref" && !n.linkRef)) return "link"
|
|
679
|
+
// The definition must still say this target.
|
|
680
|
+
const defined = documentDefs?.get(normalizeLabel(n.linkRef))
|
|
681
|
+
return documentDefs && defined !== n.link ? "link" : d
|
|
682
|
+
}
|
|
683
|
+
case "html":
|
|
684
|
+
return n.linkTag ? d : "link"
|
|
685
|
+
default:
|
|
686
|
+
return d
|
|
425
687
|
}
|
|
426
|
-
return delims.map((delim) => ({ mark: MARK_OF[delim], delim, ...(MARK_OF[delim] === "link" ? { url: n.link } : {}) }))
|
|
427
688
|
}
|
|
428
689
|
|
|
429
690
|
/**
|
|
430
|
-
* The spelling `serializeInline` gives
|
|
691
|
+
* The spelling `serializeInline` gives an inline, outermost → innermost: its `delims` hint reconciled with
|
|
431
692
|
* its marks. Editors can store it per mark and hand it back as `delims`.
|
|
432
693
|
*/
|
|
433
|
-
export function inlineDelims(n:
|
|
694
|
+
export function inlineDelims(n: Inline): InlineDelimiter[] {
|
|
434
695
|
return entriesFor(n).map((e) => e.delim)
|
|
435
696
|
}
|
|
436
697
|
|
|
437
|
-
/** The spelling
|
|
438
|
-
export function defaultInlineDelims(n:
|
|
698
|
+
/** The spelling an inline gets without a `delims` hint; a hint equal to it can be dropped. */
|
|
699
|
+
export function defaultInlineDelims(n: Inline): InlineDelimiter[] {
|
|
439
700
|
return defaultDelims(n)
|
|
440
701
|
}
|
|
441
702
|
|
|
442
|
-
const sameEntry = (a: Entry, b: Entry) =>
|
|
443
|
-
|
|
444
|
-
|
|
703
|
+
const sameEntry = (a: Entry, b: Entry) =>
|
|
704
|
+
a.delim === b.delim && a.url === b.url && a.title === b.title && a.ref === b.ref && a.tag === b.tag && a.tagEnd === b.tagEnd && a.id === b.id
|
|
705
|
+
|
|
706
|
+
/** An HTML link's opening tag, its `href` updated if the link target changed. */
|
|
707
|
+
function htmlTag(e: Entry): string {
|
|
708
|
+
const tag = e.tag!
|
|
709
|
+
const href = tag.match(/\shref="([^"]*)"/i)
|
|
710
|
+
if (!href || href[1] === e.url) return tag
|
|
711
|
+
return tag.replace(/(\shref=")[^"]*(")/i, `$1${e.url}$2`)
|
|
712
|
+
}
|
|
713
|
+
|
|
714
|
+
function opener(e: Entry): string {
|
|
715
|
+
if (e.mark !== "link") return e.delim
|
|
716
|
+
if (BRACKETED.has(e.delim)) return "["
|
|
717
|
+
if (e.delim === "<>") return "<"
|
|
718
|
+
if (e.delim === "html") return htmlTag(e)
|
|
719
|
+
return ""
|
|
720
|
+
}
|
|
721
|
+
|
|
722
|
+
/** `text`: the link text as written (unescaped), for the collapsed / shortcut forms whose label it is. */
|
|
723
|
+
function closer(e: Entry, text: string): string {
|
|
724
|
+
if (e.mark !== "link") return e.delim
|
|
725
|
+
switch (e.delim) {
|
|
726
|
+
case "link":
|
|
727
|
+
return `](${e.url}${e.title !== undefined ? ` "${e.title}"` : ""})`
|
|
728
|
+
case "ref":
|
|
729
|
+
return `][${e.ref}]`
|
|
730
|
+
case "collapsed":
|
|
731
|
+
case "shortcut": {
|
|
732
|
+
// The text is the label: if it was edited, name the label explicitly.
|
|
733
|
+
const same = normalizeLabel(unescapeLabel(e.ref ?? "")) === normalizeLabel(unescapeLabel(text))
|
|
734
|
+
return !same ? `][${e.ref}]` : e.delim === "collapsed" ? "][]" : "]"
|
|
735
|
+
}
|
|
736
|
+
case "<>":
|
|
737
|
+
return ">"
|
|
738
|
+
case "html":
|
|
739
|
+
return e.tagEnd ?? "</a>"
|
|
740
|
+
default:
|
|
741
|
+
return ""
|
|
742
|
+
}
|
|
743
|
+
}
|
|
744
|
+
|
|
745
|
+
const unescapeLabel = (s: string) => s.replace(/\\([!-/:-@[-`{-~])/g, "$1")
|
|
445
746
|
|
|
446
|
-
|
|
747
|
+
/** The fewest backticks that can fence `t`. */
|
|
748
|
+
function codeSpanTicks(t: string): number {
|
|
447
749
|
const runs = new Set((t.match(/`+/g) ?? []).map((r) => r.length))
|
|
448
750
|
let n = 1
|
|
449
751
|
while (runs.has(n)) n++
|
|
752
|
+
return n
|
|
753
|
+
}
|
|
754
|
+
|
|
755
|
+
/** `t` as a code span: `ticks` backticks when given and usable, else the fewest that work. */
|
|
756
|
+
function codeSpan(t: string, ticks?: number): string {
|
|
757
|
+
const runs = new Set((t.match(/`+/g) ?? []).map((r) => r.length))
|
|
758
|
+
const n = ticks && ticks > 0 && !runs.has(ticks) ? ticks : codeSpanTicks(t)
|
|
450
759
|
const fence = "`".repeat(n)
|
|
451
760
|
const pad = t.startsWith("`") || t.endsWith("`") ? " " : ""
|
|
452
761
|
return `${fence}${pad}${t}${pad}${fence}`
|
|
@@ -456,30 +765,44 @@ function codeSpan(t: string): string {
|
|
|
456
765
|
interface Piece {
|
|
457
766
|
s: string
|
|
458
767
|
text?: boolean
|
|
459
|
-
/** Literal text inside `[…]` link text, where
|
|
768
|
+
/** Literal text inside `[…]` link text, where unbalanced brackets must be escaped. */
|
|
460
769
|
inLink?: boolean
|
|
770
|
+
/** Characters the source escaped: always written with a backslash. */
|
|
771
|
+
forced?: boolean
|
|
772
|
+
/** Offsets (in `s`) of `|` written bare in a table cell. */
|
|
773
|
+
barePipes?: number[]
|
|
774
|
+
/** Opens / closes `[…]` link text. */
|
|
775
|
+
linkStart?: boolean
|
|
776
|
+
linkEnd?: boolean
|
|
461
777
|
}
|
|
462
778
|
|
|
463
779
|
/**
|
|
464
780
|
* Does literal `raw[i]` need a backslash? `minimal` escapes only what would otherwise parse as syntax —
|
|
465
781
|
* judged against the whole rendered line, so snake_case, `2 * 3`, `~5 min` and `a [b] c` stay as written.
|
|
466
782
|
*/
|
|
467
|
-
function needsEscape(
|
|
783
|
+
function needsEscape(r: RenderInput, i: number, minimal: boolean, defs: ReadonlyMap<string, string>) {
|
|
784
|
+
const { raw, text, bracket } = r
|
|
468
785
|
const c = raw[i]
|
|
469
786
|
switch (c) {
|
|
470
787
|
case "\\":
|
|
471
|
-
|
|
788
|
+
// A backslash ending the line (a hard break) stays as it is.
|
|
789
|
+
return !minimal || (i + 1 < raw.length && ASCII_PUNCT.test(raw[i + 1]))
|
|
472
790
|
case "`":
|
|
473
791
|
return true
|
|
474
|
-
case "[":
|
|
475
|
-
|
|
792
|
+
case "[": {
|
|
793
|
+
if (!minimal || bracket[i] === 1 || raw[i + 1] === "^") return true
|
|
794
|
+
if (/^\[(?:\\.|[^\]\\])*\][([]/.test(raw.slice(i))) return true
|
|
795
|
+
// `[label]` that a definition would turn into a shortcut link
|
|
796
|
+
const close = closingBracket(raw, i)
|
|
797
|
+
return close !== -1 && defs.has(normalizeLabel(raw.slice(i + 1, close)))
|
|
798
|
+
}
|
|
476
799
|
case "]":
|
|
477
|
-
return !minimal ||
|
|
800
|
+
return !minimal || bracket[i] === 1
|
|
478
801
|
case "@":
|
|
479
802
|
case "#":
|
|
480
803
|
case "$":
|
|
481
|
-
// Only at a chip boundary, so emails / C# / $5 survive untouched.
|
|
482
|
-
return isReferenceBoundary(raw[i - 1]) && /[A-Za-z_]/.test(raw[i + 1] ?? "")
|
|
804
|
+
// Only at a chip boundary, so emails / C# / $5 survive untouched — and not before an escaped character.
|
|
805
|
+
return isReferenceBoundary(raw[i - 1]) && /[A-Za-z_]/.test(raw[i + 1] ?? "") && !r.forced[i + 1]
|
|
483
806
|
case "*":
|
|
484
807
|
case "_":
|
|
485
808
|
case "~": {
|
|
@@ -497,100 +820,160 @@ function needsEscape(raw: string, text: Uint8Array, inLink: Uint8Array, i: numbe
|
|
|
497
820
|
return false
|
|
498
821
|
}
|
|
499
822
|
|
|
500
|
-
|
|
501
|
-
interface Rendering {
|
|
823
|
+
interface RenderInput {
|
|
502
824
|
raw: string
|
|
825
|
+
/** 1 where the character is literal text. */
|
|
826
|
+
text: Uint8Array
|
|
827
|
+
/** 1 where a literal bracket inside link text is unbalanced (so it must be escaped). */
|
|
828
|
+
bracket: Uint8Array
|
|
829
|
+
/** 1 where the source escaped the character. */
|
|
830
|
+
forced: Uint8Array
|
|
831
|
+
/** 1 where a `|` stays bare in a table cell. */
|
|
832
|
+
bare: Uint8Array
|
|
833
|
+
}
|
|
834
|
+
|
|
835
|
+
/** The rendered line: its characters, and which literal ones get a backslash. */
|
|
836
|
+
interface Rendering extends RenderInput {
|
|
503
837
|
escape: Uint8Array
|
|
504
838
|
}
|
|
505
839
|
|
|
506
|
-
function
|
|
840
|
+
function layout(pieces: Piece[]): RenderInput {
|
|
507
841
|
const raw = pieces.map((p) => p.s).join("")
|
|
508
842
|
const text = new Uint8Array(raw.length)
|
|
509
|
-
const
|
|
843
|
+
const bracket = new Uint8Array(raw.length)
|
|
844
|
+
const forced = new Uint8Array(raw.length)
|
|
845
|
+
const bare = new Uint8Array(raw.length)
|
|
510
846
|
let at = 0
|
|
847
|
+
// Brackets in link text must balance, or the link text ends early: escape the ones that don't.
|
|
848
|
+
let linkOpen: number[] | null = null
|
|
511
849
|
for (const p of pieces) {
|
|
850
|
+
if (p.linkStart) linkOpen = []
|
|
512
851
|
if (p.text) text.fill(1, at, at + p.s.length)
|
|
513
|
-
if (p.
|
|
852
|
+
if (p.forced) {
|
|
853
|
+
for (let k = 0; k < p.s.length; k++) if (ASCII_PUNCT.test(p.s[k])) forced[at + k] = 1
|
|
854
|
+
}
|
|
855
|
+
for (const k of p.barePipes ?? []) bare[at + k] = 1
|
|
856
|
+
if (linkOpen && p.text && !p.forced) {
|
|
857
|
+
for (let k = 0; k < p.s.length; k++) {
|
|
858
|
+
if (p.s[k] === "[") linkOpen.push(at + k)
|
|
859
|
+
else if (p.s[k] === "]") {
|
|
860
|
+
if (linkOpen.length) linkOpen.pop()
|
|
861
|
+
else bracket[at + k] = 1
|
|
862
|
+
}
|
|
863
|
+
}
|
|
864
|
+
}
|
|
865
|
+
if (p.linkEnd && linkOpen) {
|
|
866
|
+
for (const k of linkOpen) bracket[k] = 1
|
|
867
|
+
linkOpen = null
|
|
868
|
+
}
|
|
514
869
|
at += p.s.length
|
|
515
870
|
}
|
|
871
|
+
return { raw, text, bracket, forced, bare }
|
|
872
|
+
}
|
|
873
|
+
|
|
874
|
+
function render(input: RenderInput, minimal: boolean, defs: ReadonlyMap<string, string>): Rendering {
|
|
875
|
+
const { raw, text, forced } = input
|
|
516
876
|
const escape = new Uint8Array(raw.length)
|
|
517
877
|
for (let i = 0; i < raw.length; i++) {
|
|
518
|
-
if (
|
|
878
|
+
if (forced[i]) escape[i] = 1
|
|
879
|
+
else if (text[i] && /[\\`*_~[\]@#$]/.test(raw[i]) && needsEscape(input, i, minimal, defs)) escape[i] = 1
|
|
519
880
|
}
|
|
520
|
-
return {
|
|
881
|
+
return { ...input, escape }
|
|
521
882
|
}
|
|
522
883
|
|
|
523
|
-
|
|
884
|
+
/** The written line; `table`: inside a GFM table cell, where pipes are escaped unless written bare. */
|
|
885
|
+
const written = ({ raw, escape, bare }: Rendering, table = false) => {
|
|
524
886
|
let out = ""
|
|
525
|
-
for (let i = 0; i < raw.length; i++)
|
|
887
|
+
for (let i = 0; i < raw.length; i++) {
|
|
888
|
+
if (escape[i]) out += `\\${raw[i]}`
|
|
889
|
+
else if (table && raw[i] === "|" && !bare[i]) out += "\\|"
|
|
890
|
+
else out += raw[i]
|
|
891
|
+
}
|
|
526
892
|
return out
|
|
527
893
|
}
|
|
528
894
|
|
|
529
895
|
/**
|
|
530
896
|
* Drop backslashes the line can do without — `footnote*`, a lone backtick, `[x]` before a `(` that isn't a
|
|
531
897
|
* link. An escaped run is dropped only if the line still means `want` with every escaped run of the same
|
|
532
|
-
* character unescaped too, so paired escapes (`\*x\*`, `` \`x\` ``) keep both backslashes.
|
|
898
|
+
* character unescaped too, so paired escapes (`\*x\*`, `` \`x\` ``) keep both backslashes. The source's own
|
|
899
|
+
* escapes are never dropped.
|
|
533
900
|
*/
|
|
534
|
-
function relax(r: Rendering, want: string):
|
|
535
|
-
const original = written(r)
|
|
901
|
+
function relax(r: Rendering, want: string, ctx: Ctx): Rendering {
|
|
536
902
|
const groups: Array<{ start: number; end: number; ch: string }> = []
|
|
537
903
|
for (let i = 0; i < r.raw.length; i++) {
|
|
538
904
|
const ch = r.raw[i]
|
|
539
|
-
if (!r.escape[i] || !/[`*_~[]/.test(ch)) continue
|
|
905
|
+
if (!r.escape[i] || r.forced[i] || !/[`*_~[]/.test(ch)) continue
|
|
540
906
|
let end = i + 1
|
|
541
|
-
if (ch !== "`" && ch !== "[") while (r.escape[end] && r.raw[end] === ch) end++
|
|
907
|
+
if (ch !== "`" && ch !== "[") while (r.escape[end] && !r.forced[end] && r.raw[end] === ch) end++
|
|
542
908
|
groups.push({ start: i, end, ch })
|
|
543
909
|
i = end - 1
|
|
544
910
|
}
|
|
545
|
-
if (!groups.length || groups.length > 32) return
|
|
546
|
-
const without = (drop: (g: (typeof groups)[number]) => boolean) => {
|
|
911
|
+
if (!groups.length || groups.length > 32) return r
|
|
912
|
+
const without = (drop: (g: (typeof groups)[number]) => boolean): Rendering => {
|
|
547
913
|
const escape = r.escape.slice()
|
|
548
914
|
for (const g of groups) if (drop(g)) escape.fill(0, g.start, g.end)
|
|
549
|
-
return
|
|
915
|
+
return { ...r, escape }
|
|
550
916
|
}
|
|
551
|
-
const droppable = new Set(groups.filter((g) => meaning(
|
|
552
|
-
if (!droppable.size) return
|
|
917
|
+
const droppable = new Set(groups.filter((g) => meaning(parseWith(written(without((o) => o.ch === g.ch)), ctx)) === want))
|
|
918
|
+
if (!droppable.size) return r
|
|
553
919
|
const relaxed = without((g) => droppable.has(g))
|
|
554
|
-
return meaning(
|
|
920
|
+
return meaning(parseWith(written(relaxed), ctx)) === want ? relaxed : r
|
|
555
921
|
}
|
|
556
922
|
|
|
557
|
-
|
|
923
|
+
const marksKey = (n: Inline) => [!!n.bold, !!n.italic, !!n.strike, linkIsMark(n) ? n.link : ""]
|
|
924
|
+
|
|
925
|
+
/** What a line means, ignoring spelling: merged text runs with their marks, plus atoms with theirs. */
|
|
558
926
|
function meaning(nodes: Inline[]): string {
|
|
559
927
|
const parts: unknown[] = []
|
|
560
928
|
let last: { key: string; text: string } | undefined
|
|
561
929
|
for (const n of nodes) {
|
|
562
930
|
if (n.type !== "text") {
|
|
563
931
|
last = undefined
|
|
564
|
-
parts.push(
|
|
932
|
+
parts.push(
|
|
933
|
+
n.type === "image"
|
|
934
|
+
? ["img", n.src, linkIsMark(n) ? "" : (n.link ?? ""), ...marksKey(n)]
|
|
935
|
+
: ["ref", n.kind, n.id, ...marksKey(n)],
|
|
936
|
+
)
|
|
565
937
|
continue
|
|
566
938
|
}
|
|
567
939
|
if (!n.text) continue
|
|
568
|
-
const key = JSON.stringify([
|
|
940
|
+
const key = JSON.stringify([...marksKey(n), !!n.code])
|
|
569
941
|
if (last && last.key === key && !n.code) last.text += n.text
|
|
570
942
|
else parts.push((last = { key, text: n.text }))
|
|
571
943
|
}
|
|
572
944
|
return JSON.stringify(parts)
|
|
573
945
|
}
|
|
574
946
|
|
|
575
|
-
|
|
576
|
-
|
|
577
|
-
|
|
578
|
-
|
|
579
|
-
|
|
947
|
+
export interface SerializeInlineOptions {
|
|
948
|
+
/** Inside a GFM table cell: pipes are written `\|` (except those a code span recorded as bare). */
|
|
949
|
+
tableCell?: boolean
|
|
950
|
+
}
|
|
951
|
+
|
|
952
|
+
// Serializes inline nodes back to markdown, as written: each inline's `delims` restores the author's markers,
|
|
953
|
+
// nesting and link form, and formatting shared by adjacent inlines (chips and images included) is opened once
|
|
954
|
+
// — `**bold [link](url) inside**`, `**ask @brett**` and `[**a** b](url)` come back whole instead of per run.
|
|
955
|
+
// Text is escaped minimally (plus every escape the source wrote); if the minimal form would parse back
|
|
956
|
+
// differently, everything escapable is escaped instead.
|
|
957
|
+
export function serializeInline(nodes: Inline[], options: SerializeInlineOptions = {}): string {
|
|
580
958
|
const pieces: Piece[] = []
|
|
581
|
-
const open: Entry
|
|
959
|
+
const open: Array<Entry & { start: number }> = []
|
|
960
|
+
// Reference forms resolve through definitions: the document's, else the ones the links themselves imply.
|
|
961
|
+
const defs = new Map(documentDefs ?? [])
|
|
582
962
|
const closeTo = (depth: number) => {
|
|
583
|
-
while (open.length > depth)
|
|
963
|
+
while (open.length > depth) {
|
|
964
|
+
const e = open.pop()!
|
|
965
|
+
const text = pieces.slice(e.start + 1).map((p) => p.s).join("")
|
|
966
|
+
pieces.push({ s: closer(e, text), ...(BRACKETED.has(e.delim) ? { linkEnd: true } : {}) })
|
|
967
|
+
}
|
|
584
968
|
}
|
|
585
969
|
for (const n of nodes) {
|
|
586
|
-
if (n.type
|
|
587
|
-
closeTo(0)
|
|
588
|
-
pieces.push({ s: n.type === "image" ? serializeImage(n) : serializeReference(n) })
|
|
589
|
-
continue
|
|
590
|
-
}
|
|
591
|
-
if (!n.text) continue
|
|
970
|
+
if (n.type === "text" && !n.text && !linkIsMark(n)) continue
|
|
592
971
|
const wanted = entriesFor(n)
|
|
593
|
-
|
|
972
|
+
for (const e of wanted) {
|
|
973
|
+
if (e.ref !== undefined && e.url !== undefined && !defs.has(normalizeLabel(e.ref))) defs.set(normalizeLabel(e.ref), e.url)
|
|
974
|
+
}
|
|
975
|
+
if (n.type === "image" && n.srcRef !== undefined && !defs.has(normalizeLabel(n.srcRef))) defs.set(normalizeLabel(n.srcRef), n.src)
|
|
976
|
+
// Keep the open formatting this inline still wants (outermost first), close the rest, open what's missing.
|
|
594
977
|
const missing = [...wanted]
|
|
595
978
|
let keep = 0
|
|
596
979
|
for (; keep < open.length; keep++) {
|
|
@@ -600,22 +983,32 @@ export function serializeInline(nodes: Inline[]): string {
|
|
|
600
983
|
}
|
|
601
984
|
closeTo(keep)
|
|
602
985
|
for (const w of missing) {
|
|
603
|
-
open.push(w)
|
|
604
|
-
pieces.push({ s: opener(w) })
|
|
986
|
+
open.push({ ...w, start: pieces.length })
|
|
987
|
+
pieces.push({ s: opener(w), ...(BRACKETED.has(w.delim) ? { linkStart: true } : {}) })
|
|
605
988
|
}
|
|
989
|
+
const inLink = open.some((e) => BRACKETED.has(e.delim))
|
|
606
990
|
const innermost = open[open.length - 1]
|
|
607
|
-
if (n.
|
|
608
|
-
else if (
|
|
609
|
-
else
|
|
991
|
+
if (n.type === "image") pieces.push({ s: serializeImage(n, !linkIsMark(n)) })
|
|
992
|
+
else if (n.type === "reference") pieces.push({ s: serializeReference(n) })
|
|
993
|
+
else if (n.code) {
|
|
994
|
+
const s = codeSpan(n.text, n.ticks)
|
|
995
|
+
const pad = s.indexOf(n.text)
|
|
996
|
+
pieces.push({ s, ...(n.barePipes?.length ? { barePipes: n.barePipes.map((k) => k + pad) } : {}) })
|
|
997
|
+
} else if (n.escaped) pieces.push({ s: n.text, text: true, forced: true, inLink })
|
|
998
|
+
else if (innermost?.mark === "link" && (innermost.delim === "url" || innermost.delim === "<>")) pieces.push({ s: n.text }) // autolink
|
|
999
|
+
else pieces.push({ s: n.text, text: true, inLink })
|
|
610
1000
|
}
|
|
611
1001
|
closeTo(0)
|
|
612
1002
|
|
|
613
|
-
const
|
|
614
|
-
|
|
1003
|
+
const ctx: Ctx = { defs }
|
|
1004
|
+
const table = !!options.tableCell
|
|
1005
|
+
const input = layout(pieces)
|
|
1006
|
+
const minimal = render(input, true, defs)
|
|
1007
|
+
if (!pieces.some((p) => p.text)) return written(minimal, table)
|
|
615
1008
|
const want = meaning(nodes)
|
|
616
|
-
if (meaning(
|
|
617
|
-
const full =
|
|
618
|
-
return meaning(
|
|
1009
|
+
if (meaning(parseWith(written(minimal), ctx)) === want) return written(relax(minimal, want, ctx), table)
|
|
1010
|
+
const full = render(input, false, defs)
|
|
1011
|
+
return meaning(parseWith(written(full), ctx)) === want ? written(full, table) : written(minimal, table)
|
|
619
1012
|
}
|
|
620
1013
|
|
|
621
1014
|
export function plainText(nodes: Inline[]): string {
|