@brett_lamy/docstream 1.2.1 → 1.2.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +36 -5
- package/package.json +1 -1
- package/src/demo/DemoViewer.tsx +52 -9
- package/src/demo/index.ts +1 -0
- package/src/demo/markdown.ts +17 -3
- package/src/demo/surface.ts +67 -0
- package/src/gitbook/ast.ts +43 -2
- package/src/gitbook/attrs.ts +39 -0
- package/src/gitbook/flatten.ts +1 -1
- package/src/gitbook/index.ts +2 -1
- package/src/gitbook/inline.ts +502 -163
- package/src/gitbook/parse.ts +43 -22
- package/src/gitbook/serialize.ts +52 -43
- package/src/index.ts +2 -0
- package/src/styles.css +29 -19
package/src/gitbook/inline.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import type { Inline, InlineImageNode, ReferenceNode, TextNode } from "./ast"
|
|
1
|
+
import type { Inline, InlineDelimiter, InlineImageNode, ReferenceNode, TextNode } from "./ast"
|
|
2
2
|
|
|
3
3
|
type Marks = Partial<Omit<TextNode, "type" | "text">>
|
|
4
4
|
|
|
@@ -17,10 +17,10 @@ const REFERENCE_BODY_RE = /^[@#]([A-Za-z_][\w.-]*)/
|
|
|
17
17
|
const CODEBASE_BODY_RE = /^\$([A-Za-z_][\w.\/-]*)/
|
|
18
18
|
const SIGIL_KIND = { "@": "mention", "#": "tag", $: "codebase" } as const
|
|
19
19
|
const KIND_SIGIL = { mention: "@", tag: "#", codebase: "$" } as const
|
|
20
|
-
// Chips are only recognized at start-of-input or after whitespace
|
|
20
|
+
// Chips are only recognized at start-of-input or after whitespace, open brackets or an emphasis marker,
|
|
21
21
|
// so brett@replay.io and C# stay plain text.
|
|
22
22
|
const isReferenceBoundary = (prev: string | undefined) =>
|
|
23
|
-
prev === undefined || /[\s([{]/.test(prev)
|
|
23
|
+
prev === undefined || /[\s([{*_~]/.test(prev)
|
|
24
24
|
|
|
25
25
|
function imgAttrs(attrStr: string): Omit<InlineImageNode, "type" | "link"> {
|
|
26
26
|
const attr = (name: string) => attrStr.match(new RegExp(`${name}="([^"]*)"`, "i"))?.[1]
|
|
@@ -34,182 +34,342 @@ function imgAttrs(attrStr: string): Omit<InlineImageNode, "type" | "link"> {
|
|
|
34
34
|
return out
|
|
35
35
|
}
|
|
36
36
|
|
|
37
|
-
//
|
|
38
|
-
//
|
|
39
|
-
//
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
let buf = ""
|
|
37
|
+
// ─── Parsing ─────────────────────────────────────────────────────────────────────────────────────────────
|
|
38
|
+
// Two passes, after CommonMark: tokenize (escapes, code spans, links, images, chips, autolinks and `*` / `_` /
|
|
39
|
+
// `~~` delimiter runs with their flanking), then pair the delimiter runs with the CommonMark "process
|
|
40
|
+
// emphasis" algorithm (rule of 3 included), which yields a tree. The tree is flattened into TextNodes; each
|
|
41
|
+
// remembers how its formatting was spelled (`delims`) whenever that differs from the default output.
|
|
43
42
|
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
43
|
+
type Mark = "bold" | "italic" | "strike" | "link"
|
|
44
|
+
|
|
45
|
+
type Tree =
|
|
46
|
+
| { k: "text"; text: string }
|
|
47
|
+
| { k: "code"; text: string }
|
|
48
|
+
| { k: "atom"; node: InlineImageNode | ReferenceNode }
|
|
49
|
+
| { k: "wrap"; delim: InlineDelimiter; url?: string; children: Tree[] }
|
|
50
|
+
|
|
51
|
+
interface DelimRun {
|
|
52
|
+
k: "delim"
|
|
53
|
+
ch: "*" | "_" | "~"
|
|
54
|
+
/** Characters still unmatched. */
|
|
55
|
+
n: number
|
|
56
|
+
/** Length of the run as written (rule of 3). */
|
|
57
|
+
orig: number
|
|
58
|
+
open: boolean
|
|
59
|
+
close: boolean
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
type Tok = Tree | DelimRun
|
|
63
|
+
|
|
64
|
+
const MARK_OF: Record<InlineDelimiter, Mark> = {
|
|
65
|
+
"**": "bold",
|
|
66
|
+
__: "bold",
|
|
67
|
+
"*": "italic",
|
|
68
|
+
_: "italic",
|
|
69
|
+
"~~": "strike",
|
|
70
|
+
link: "link",
|
|
71
|
+
"<>": "link",
|
|
72
|
+
url: "link",
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
const ASCII_PUNCT = /[!-/:-@[-`{-~]/
|
|
76
|
+
const isSpace = (c: string | undefined) => c === undefined || /\s/.test(c)
|
|
77
|
+
const isPunct = (c: string | undefined) => c !== undefined && /[\p{P}\p{S}]/u.test(c)
|
|
78
|
+
|
|
79
|
+
/** CommonMark left-/right-flanking for a delimiter run between `prev` and `next` (undefined = line edge). */
|
|
80
|
+
function flanking(prev: string | undefined, next: string | undefined) {
|
|
81
|
+
return {
|
|
82
|
+
left: !isSpace(next) && (!isPunct(next) || isSpace(prev) || isPunct(prev)),
|
|
83
|
+
right: !isSpace(prev) && (!isPunct(prev) || isSpace(next) || isPunct(next)),
|
|
84
|
+
}
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
/** Can a run of `ch` between `prev` and `next` open / close emphasis? */
|
|
88
|
+
function openClose(ch: string, prev: string | undefined, next: string | undefined) {
|
|
89
|
+
const { left, right } = flanking(prev, next)
|
|
90
|
+
if (ch !== "_") return { open: left, close: right }
|
|
91
|
+
// `_` never opens or closes intraword: snake_case stays literal.
|
|
92
|
+
return { open: left && (!right || isPunct(prev)), close: right && (!left || isPunct(next)) }
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
/** Start of the next backtick run of exactly `n` at or after `from`, or -1. */
|
|
96
|
+
function closingBackticks(src: string, from: number, n: number): number {
|
|
97
|
+
let j = from
|
|
98
|
+
while (j < src.length) {
|
|
99
|
+
const k = src.indexOf("`", j)
|
|
100
|
+
if (k === -1) return -1
|
|
101
|
+
let m = 1
|
|
102
|
+
while (src[k + m] === "`") m++
|
|
103
|
+
if (m === n) return k
|
|
104
|
+
j = k + m
|
|
49
105
|
}
|
|
106
|
+
return -1
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
const LINK_RE = /^\[((?:\\.|[^\]\\])*)\]\(([^)\s]+)\)/
|
|
110
|
+
const REF_LINK_RE = /^\[((?:\\.|[^\]\\])*)\]\[([^\]]*)\]/
|
|
111
|
+
|
|
112
|
+
/** `inLink`: tokenizing link text, where CommonMark allows no further links (or autolinks). */
|
|
113
|
+
function tokenize(src: string, inLink = false): Tok[] {
|
|
114
|
+
const toks: Tok[] = []
|
|
115
|
+
let buf = ""
|
|
116
|
+
const push = (t: Tok) => {
|
|
117
|
+
if (buf) toks.push({ k: "text", text: buf })
|
|
118
|
+
buf = ""
|
|
119
|
+
toks.push(t)
|
|
120
|
+
}
|
|
121
|
+
const wrap = (delim: InlineDelimiter, url: string, inner: string | Tree[]): Tok => ({
|
|
122
|
+
k: "wrap",
|
|
123
|
+
delim,
|
|
124
|
+
url,
|
|
125
|
+
children: typeof inner === "string" ? pairEmphasis(tokenize(inner, true)) : inner,
|
|
126
|
+
})
|
|
50
127
|
|
|
51
128
|
let i = 0
|
|
52
129
|
while (i < src.length) {
|
|
130
|
+
const c = src[i]
|
|
53
131
|
const rest = src.slice(i)
|
|
54
132
|
|
|
55
|
-
|
|
56
|
-
|
|
133
|
+
// Backslash escapes: ASCII punctuation only; `C:\path` keeps its backslash.
|
|
134
|
+
if (c === "\\" && i + 1 < src.length && ASCII_PUNCT.test(src[i + 1])) {
|
|
135
|
+
buf += src[i + 1]
|
|
57
136
|
i += 2
|
|
58
137
|
continue
|
|
59
138
|
}
|
|
60
139
|
|
|
61
|
-
//
|
|
62
|
-
if (
|
|
63
|
-
|
|
140
|
+
// Code span: a backtick run closed by the next run of the same length. Content is kept verbatim.
|
|
141
|
+
if (c === "`") {
|
|
142
|
+
let n = 1
|
|
143
|
+
while (src[i + n] === "`") n++
|
|
144
|
+
const end = closingBackticks(src, i + n, n)
|
|
64
145
|
if (end !== -1) {
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
146
|
+
push({ k: "code", text: src.slice(i + n, end) })
|
|
147
|
+
i = end + n
|
|
148
|
+
} else {
|
|
149
|
+
buf += src.slice(i, i + n)
|
|
150
|
+
i += n
|
|
69
151
|
}
|
|
70
|
-
}
|
|
71
|
-
|
|
72
|
-
// <a href="…"><img …></a> — linked image (GitHub badge style)
|
|
73
|
-
const linkedImg = rest.match(/^<a\s[^>]*href="([^"]*)"[^>]*>\s*<img\s([^>]*?)\/?>\s*<\/a>/i)
|
|
74
|
-
if (linkedImg) {
|
|
75
|
-
flush()
|
|
76
|
-
out.push({ type: "image", ...imgAttrs(linkedImg[2]), link: linkedImg[1] })
|
|
77
|
-
i += linkedImg[0].length
|
|
78
152
|
continue
|
|
79
153
|
}
|
|
80
154
|
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
155
|
+
if (c === "<") {
|
|
156
|
+
// <a href="…"><img …></a> — linked image (GitHub badge style)
|
|
157
|
+
const linkedImg = rest.match(/^<a\s[^>]*href="([^"]*)"[^>]*>\s*<img\s([^>]*?)\/?>\s*<\/a>/i)
|
|
158
|
+
if (linkedImg) {
|
|
159
|
+
push({ k: "atom", node: { type: "image", ...imgAttrs(linkedImg[2]), link: linkedImg[1] } })
|
|
160
|
+
i += linkedImg[0].length
|
|
161
|
+
continue
|
|
162
|
+
}
|
|
163
|
+
// <img …> — bare inline image
|
|
164
|
+
const htmlImg = rest.match(/^<img\s([^>]*?)\/?>/i)
|
|
165
|
+
if (htmlImg) {
|
|
166
|
+
push({ k: "atom", node: { type: "image", ...imgAttrs(htmlImg[1]) } })
|
|
167
|
+
i += htmlImg[0].length
|
|
168
|
+
continue
|
|
169
|
+
}
|
|
170
|
+
// <a href="…">text</a> — html link
|
|
171
|
+
const htmlLink = !inLink && rest.match(/^<a\s[^>]*href="([^"]*)"[^>]*>(.*?)<\/a>/i)
|
|
172
|
+
if (htmlLink) {
|
|
173
|
+
push(wrap("link", htmlLink[1], htmlLink[2]))
|
|
174
|
+
i += htmlLink[0].length
|
|
175
|
+
continue
|
|
176
|
+
}
|
|
177
|
+
// <https://…> — angle-bracket autolink
|
|
178
|
+
const angleLink = !inLink && rest.match(/^<(https?:\/\/[^>\s]+)>/i)
|
|
179
|
+
if (angleLink) {
|
|
180
|
+
push(wrap("<>", angleLink[1], [{ k: "text", text: angleLink[1] }]))
|
|
181
|
+
i += angleLink[0].length
|
|
182
|
+
continue
|
|
183
|
+
}
|
|
97
184
|
}
|
|
98
185
|
|
|
99
186
|
//  — markdown inline image
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
}
|
|
107
|
-
|
|
108
|
-
const delims: Array<[string, Marks]> = [
|
|
109
|
-
["**", { bold: true }],
|
|
110
|
-
["~~", { strike: true }],
|
|
111
|
-
["*", { italic: true }],
|
|
112
|
-
["_", { italic: true }],
|
|
113
|
-
]
|
|
114
|
-
let matched = false
|
|
115
|
-
for (const [d, mark] of delims) {
|
|
116
|
-
if (rest.startsWith(d)) {
|
|
117
|
-
const end = src.indexOf(d, i + d.length)
|
|
118
|
-
if (end > i + d.length - 1 && end !== -1 && src.slice(i + d.length, end).length > 0) {
|
|
119
|
-
flush()
|
|
120
|
-
out.push(...parseInline(src.slice(i + d.length, end), { ...marks, ...mark }))
|
|
121
|
-
i = end + d.length
|
|
122
|
-
matched = true
|
|
123
|
-
break
|
|
124
|
-
}
|
|
187
|
+
if (c === "!") {
|
|
188
|
+
const mdImg = rest.match(/^!\[([^\]]*)\]\(([^)\s]+)\)/)
|
|
189
|
+
if (mdImg) {
|
|
190
|
+
push({ k: "atom", node: { type: "image", src: mdImg[2], ...(mdImg[1] ? { alt: mdImg[1] } : {}) } })
|
|
191
|
+
i += mdImg[0].length
|
|
192
|
+
continue
|
|
125
193
|
}
|
|
126
194
|
}
|
|
127
|
-
if (matched) continue
|
|
128
|
-
|
|
129
|
-
// [^id] — footnote citation marker (must precede [text](url) / [text][ref])
|
|
130
|
-
const cite = rest.match(/^\[\^([^\]\s]+)\]/)
|
|
131
|
-
if (cite) {
|
|
132
|
-
flush()
|
|
133
|
-
const def = footnoteDefinitions.get(cite[1])
|
|
134
|
-
out.push({ type: "reference", kind: "citation", id: cite[1], ...def })
|
|
135
|
-
i += cite[0].length
|
|
136
|
-
continue
|
|
137
|
-
}
|
|
138
195
|
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
const
|
|
142
|
-
if (
|
|
143
|
-
const
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
i += 1 + id.length
|
|
148
|
-
continue
|
|
149
|
-
}
|
|
196
|
+
if (c === "[") {
|
|
197
|
+
// [^id] — footnote citation marker
|
|
198
|
+
const cite = rest.match(/^\[\^([^\]\s]+)\]/)
|
|
199
|
+
if (cite) {
|
|
200
|
+
const def = footnoteDefinitions.get(cite[1])
|
|
201
|
+
push({ k: "atom", node: { type: "reference", kind: "citation", id: cite[1], ...def } })
|
|
202
|
+
i += cite[0].length
|
|
203
|
+
continue
|
|
150
204
|
}
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
// [text](url)
|
|
154
|
-
if (rest[0] === "[") {
|
|
155
|
-
const m = rest.match(/^\[([^\]]*)\]\(([^)\s]+)\)/)
|
|
205
|
+
// [text](url) — link text is its own emphasis scope, as in CommonMark
|
|
206
|
+
const m = !inLink && rest.match(LINK_RE)
|
|
156
207
|
if (m) {
|
|
157
|
-
|
|
158
|
-
const inner = marks.bold || marks.italic || marks.strike ? { linkInner: true as const } : {}
|
|
159
|
-
out.push(...parseInline(m[1], { ...marks, link: m[2], ...inner }))
|
|
208
|
+
push(wrap("link", m[2], m[1]))
|
|
160
209
|
i += m[0].length
|
|
161
210
|
continue
|
|
162
211
|
}
|
|
163
212
|
// [text][ref] / [text][] — reference-style links
|
|
164
|
-
const ref = rest.match(
|
|
213
|
+
const ref = !inLink && rest.match(REF_LINK_RE)
|
|
165
214
|
if (ref) {
|
|
166
|
-
const
|
|
167
|
-
const url = refDefinitions.get(key)
|
|
215
|
+
const url = refDefinitions.get((ref[2] || ref[1]).toLowerCase())
|
|
168
216
|
if (url) {
|
|
169
|
-
|
|
170
|
-
out.push(...parseInline(ref[1], { ...marks, link: url }))
|
|
217
|
+
push(wrap("link", url, ref[1]))
|
|
171
218
|
i += ref[0].length
|
|
172
219
|
continue
|
|
173
220
|
}
|
|
174
221
|
}
|
|
175
222
|
}
|
|
176
223
|
|
|
177
|
-
//
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
224
|
+
// @mention / #tag / $codebase chips, only at a word boundary
|
|
225
|
+
if ((c === "@" || c === "#" || c === "$") && isReferenceBoundary(src[i - 1])) {
|
|
226
|
+
const m = rest.match(c === "$" ? CODEBASE_BODY_RE : REFERENCE_BODY_RE)
|
|
227
|
+
if (m) {
|
|
228
|
+
const id = m[1].replace(/[./-]+$/, "")
|
|
229
|
+
if (id) {
|
|
230
|
+
push({ k: "atom", node: { type: "reference", kind: SIGIL_KIND[c], id } })
|
|
231
|
+
i += 1 + id.length
|
|
232
|
+
continue
|
|
233
|
+
}
|
|
234
|
+
}
|
|
184
235
|
}
|
|
185
236
|
|
|
186
|
-
// bare URL autolink (GFM) — at start or
|
|
187
|
-
if (
|
|
188
|
-
/^https?:\/\//i.test(rest) &&
|
|
189
|
-
(buf === "" || /\s$/.test(buf))
|
|
190
|
-
) {
|
|
237
|
+
// bare URL autolink (GFM) — at start, after whitespace or an emphasis marker
|
|
238
|
+
if (!inLink && (c === "h" || c === "H") && /^https?:\/\//i.test(rest) && (i === 0 || /[\s*_~]/.test(src[i - 1]))) {
|
|
191
239
|
const m = rest.match(/^https?:\/\/[^\s<>"')\]]+/i)!
|
|
192
|
-
const url = m[0].replace(/[
|
|
193
|
-
|
|
194
|
-
out.push({ type: "text", text: url, ...marks, link: url })
|
|
240
|
+
const url = m[0].replace(/[.,;:!?*_~]+$/, "")
|
|
241
|
+
push(wrap("url", url, [{ k: "text", text: url }]))
|
|
195
242
|
i += url.length
|
|
196
243
|
continue
|
|
197
244
|
}
|
|
198
245
|
|
|
199
|
-
|
|
246
|
+
// Delimiter runs: any run of `*` / `_`, exactly two `~`.
|
|
247
|
+
if (c === "*" || c === "_" || (c === "~" && src[i + 1] === "~")) {
|
|
248
|
+
let n = 1
|
|
249
|
+
while (src[i + n] === c) n++
|
|
250
|
+
if (c === "~" && n !== 2) {
|
|
251
|
+
buf += src.slice(i, i + n)
|
|
252
|
+
i += n
|
|
253
|
+
continue
|
|
254
|
+
}
|
|
255
|
+
push({ k: "delim", ch: c, n, orig: n, ...openClose(c, src[i - 1], src[i + n]) })
|
|
256
|
+
i += n
|
|
257
|
+
continue
|
|
258
|
+
}
|
|
259
|
+
|
|
260
|
+
buf += c
|
|
200
261
|
i++
|
|
201
262
|
}
|
|
202
|
-
|
|
263
|
+
if (buf) toks.push({ k: "text", text: buf })
|
|
264
|
+
return toks
|
|
265
|
+
}
|
|
266
|
+
|
|
267
|
+
/** Leftover delimiter runs become literal text; adjacent text merges. */
|
|
268
|
+
function settle(toks: Tok[]): Tree[] {
|
|
269
|
+
const out: Tree[] = []
|
|
270
|
+
for (const t of toks) {
|
|
271
|
+
const tree: Tree = t.k === "delim" ? { k: "text", text: t.ch.repeat(t.n) } : t
|
|
272
|
+
const last = out[out.length - 1]
|
|
273
|
+
if (tree.k === "text" && last?.k === "text") out[out.length - 1] = { k: "text", text: last.text + tree.text }
|
|
274
|
+
else if (tree.k !== "text" || tree.text) out.push(tree)
|
|
275
|
+
}
|
|
276
|
+
return out
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
/** CommonMark "process emphasis": pair closers with the nearest eligible opener, innermost first. */
|
|
280
|
+
function pairEmphasis(toks: Tok[]): Tree[] {
|
|
281
|
+
let ci = 0
|
|
282
|
+
while (ci < toks.length) {
|
|
283
|
+
const closer = toks[ci]
|
|
284
|
+
if (closer.k !== "delim" || !closer.close || closer.n === 0) {
|
|
285
|
+
ci++
|
|
286
|
+
continue
|
|
287
|
+
}
|
|
288
|
+
let oi = ci - 1
|
|
289
|
+
for (; oi >= 0; oi--) {
|
|
290
|
+
const o = toks[oi]
|
|
291
|
+
if (o.k !== "delim" || o.ch !== closer.ch || !o.open || o.n === 0) continue
|
|
292
|
+
if (closer.ch === "~") {
|
|
293
|
+
if (o.n === 2 && closer.n === 2) break
|
|
294
|
+
continue
|
|
295
|
+
}
|
|
296
|
+
// Rule of 3: a run that can both open and close can't pair with one whose lengths sum to a multiple of 3.
|
|
297
|
+
const both = o.close || closer.open
|
|
298
|
+
if (both && (o.orig + closer.orig) % 3 === 0 && !(o.orig % 3 === 0 && closer.orig % 3 === 0)) continue
|
|
299
|
+
break
|
|
300
|
+
}
|
|
301
|
+
if (oi < 0) {
|
|
302
|
+
ci++
|
|
303
|
+
continue
|
|
304
|
+
}
|
|
305
|
+
const opener = toks[oi] as DelimRun
|
|
306
|
+
const use = closer.ch === "~" ? 2 : opener.n >= 2 && closer.n >= 2 ? 2 : 1
|
|
307
|
+
const delim = (closer.ch === "~" ? "~~" : closer.ch.repeat(use)) as InlineDelimiter
|
|
308
|
+
const node: Tree = { k: "wrap", delim, children: settle(toks.slice(oi + 1, ci)) }
|
|
309
|
+
opener.n -= use
|
|
310
|
+
closer.n -= use
|
|
311
|
+
const replaced: Tok[] = [...(opener.n > 0 ? [opener] : []), node, ...(closer.n > 0 ? [closer] : [])]
|
|
312
|
+
toks.splice(oi, ci - oi + 1, ...replaced)
|
|
313
|
+
// Continue with the closer's remainder (it may close an outer opener too), or what follows it.
|
|
314
|
+
ci = oi + replaced.length - (closer.n > 0 ? 1 : 0)
|
|
315
|
+
}
|
|
316
|
+
return settle(toks)
|
|
317
|
+
}
|
|
318
|
+
|
|
319
|
+
/** The spelling a run gets when nothing says otherwise — what serialization produced before `delims`. */
|
|
320
|
+
function defaultDelims(n: TextNode): InlineDelimiter[] {
|
|
321
|
+
const emphasis: InlineDelimiter[] = []
|
|
322
|
+
if (n.strike) emphasis.push("~~")
|
|
323
|
+
if (n.italic) emphasis.push("_")
|
|
324
|
+
if (n.bold) emphasis.push("**")
|
|
325
|
+
if (!n.link) return emphasis
|
|
326
|
+
const inner = !!n.linkInner && emphasis.length > 0
|
|
327
|
+
const form: InlineDelimiter = (inner || !emphasis.length) && n.text === n.link && !n.code ? "url" : "link"
|
|
328
|
+
return inner ? [...emphasis, form] : [form, ...emphasis]
|
|
329
|
+
}
|
|
330
|
+
|
|
331
|
+
const sameList = (a: readonly string[], b: readonly string[]) => a.length === b.length && a.every((x, k) => x === b[k])
|
|
332
|
+
|
|
333
|
+
function flatten(trees: Tree[], path: Array<{ delim: InlineDelimiter; url?: string }>, base: Marks, out: Inline[]) {
|
|
334
|
+
for (const t of trees) {
|
|
335
|
+
if (t.k === "wrap") {
|
|
336
|
+
flatten(t.children, [...path, { delim: t.delim, url: t.url }], base, out)
|
|
337
|
+
continue
|
|
338
|
+
}
|
|
339
|
+
if (t.k === "atom") {
|
|
340
|
+
out.push(t.node)
|
|
341
|
+
continue
|
|
342
|
+
}
|
|
343
|
+
const n: TextNode = { type: "text", text: t.text, ...base }
|
|
344
|
+
let emphasis = false
|
|
345
|
+
for (const e of path) {
|
|
346
|
+
const mark = MARK_OF[e.delim]
|
|
347
|
+
if (mark === "link") {
|
|
348
|
+
n.link = e.url
|
|
349
|
+
if (emphasis) n.linkInner = true
|
|
350
|
+
} else {
|
|
351
|
+
n[mark] = true
|
|
352
|
+
emphasis = true
|
|
353
|
+
}
|
|
354
|
+
}
|
|
355
|
+
if (t.k === "code") n.code = true
|
|
356
|
+
const delims = path.map((e) => e.delim)
|
|
357
|
+
if (!sameList(delims, defaultDelims(n))) n.delims = delims
|
|
358
|
+
out.push(n)
|
|
359
|
+
}
|
|
360
|
+
}
|
|
361
|
+
|
|
362
|
+
// Parses GitBook/GFM inline markdown into flat TextNodes with marks.
|
|
363
|
+
// Supported: **bold** / __bold__, _italic_ / *italic*, ***both***, ~~strike~~, `code`, [text](url),
|
|
364
|
+
// [text][ref], <url> and bare URLs, @mention / #tag / $codebase / [^n] chips, ,
|
|
365
|
+
// plus GitHub-style inline HTML: <img …>, <a href><img …></a>, <a href>text</a>.
|
|
366
|
+
export function parseInline(src: string, marks: Marks = {}): Inline[] {
|
|
367
|
+
const out: Inline[] = []
|
|
368
|
+
flatten(pairEmphasis(tokenize(src)), [], marks, out)
|
|
203
369
|
return out
|
|
204
370
|
}
|
|
205
371
|
|
|
206
|
-
|
|
207
|
-
t
|
|
208
|
-
.replace(/([*_~`[\]\\])/g, "\\$1")
|
|
209
|
-
// Escape @/#/$ only at a chip boundary so emails/C#/$5 survive untouched.
|
|
210
|
-
// Note: node-local — a boundary formed across adjacent inline nodes
|
|
211
|
-
// (previous node ending in whitespace) is not caught; rare, accepted.
|
|
212
|
-
.replace(/(^|[\s([{])([@#$])(?=[A-Za-z_])/g, "$1\\$2")
|
|
372
|
+
// ─── Serializing ─────────────────────────────────────────────────────────────────────────────────────────
|
|
213
373
|
|
|
214
374
|
export function serializeReference(n: ReferenceNode): string {
|
|
215
375
|
if (n.kind === "citation") return `[^${n.id}]`
|
|
@@ -229,54 +389,233 @@ function serializeImage(n: InlineImageNode): string {
|
|
|
229
389
|
return n.link ? `<a href="${n.link}">${img}</a>` : img
|
|
230
390
|
}
|
|
231
391
|
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
const MARKER: Record<Emphasis, string> = { strike: "~~", italic: "_", bold: "**" }
|
|
392
|
+
interface Entry {
|
|
393
|
+
mark: Mark
|
|
394
|
+
delim: InlineDelimiter
|
|
395
|
+
url?: string
|
|
396
|
+
}
|
|
238
397
|
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
398
|
+
/** The run's formatting, outermost → innermost: its `delims` hint reconciled with its actual marks. */
|
|
399
|
+
function entriesFor(n: TextNode): Entry[] {
|
|
400
|
+
const fallback = defaultDelims(n)
|
|
401
|
+
let delims = fallback
|
|
402
|
+
if (n.delims?.length) {
|
|
403
|
+
const has = (m: Mark) => (m === "link" ? !!n.link : !!n[m])
|
|
404
|
+
delims = []
|
|
405
|
+
for (const d of n.delims) {
|
|
406
|
+
const m = MARK_OF[d]
|
|
407
|
+
if (m && has(m) && !(m === "link" && delims.some((x) => MARK_OF[x] === "link"))) delims.push(d)
|
|
408
|
+
}
|
|
409
|
+
// Marks without a spelling: emphasis innermost (inside a trailing link), the link where it defaults to.
|
|
410
|
+
for (const d of fallback) {
|
|
411
|
+
const m = MARK_OF[d]
|
|
412
|
+
if (delims.some((x) => MARK_OF[x] === m)) continue
|
|
413
|
+
if (m === "link") {
|
|
414
|
+
if (n.linkInner) delims.push(d)
|
|
415
|
+
else delims.unshift(d)
|
|
416
|
+
} else {
|
|
417
|
+
const last = delims[delims.length - 1]
|
|
418
|
+
delims.splice(last && MARK_OF[last] === "link" ? delims.length - 1 : delims.length, 0, d)
|
|
419
|
+
}
|
|
420
|
+
}
|
|
421
|
+
// An autolink shows its URL as its text: only valid innermost, around an unchanged URL.
|
|
422
|
+
delims = delims.map((d, k) =>
|
|
423
|
+
(d === "url" || d === "<>") && (k !== delims.length - 1 || n.text !== n.link || n.code) ? "link" : d,
|
|
424
|
+
)
|
|
425
|
+
}
|
|
426
|
+
return delims.map((delim) => ({ mark: MARK_OF[delim], delim, ...(MARK_OF[delim] === "link" ? { url: n.link } : {}) }))
|
|
242
427
|
}
|
|
243
428
|
|
|
244
|
-
|
|
429
|
+
/**
|
|
430
|
+
* The spelling `serializeInline` gives a text run, outermost → innermost: its `delims` hint reconciled with
|
|
431
|
+
* its marks. Editors can store it per mark and hand it back as `delims`.
|
|
432
|
+
*/
|
|
433
|
+
export function inlineDelims(n: TextNode): InlineDelimiter[] {
|
|
434
|
+
return entriesFor(n).map((e) => e.delim)
|
|
435
|
+
}
|
|
436
|
+
|
|
437
|
+
/** The spelling a text run gets without a `delims` hint; a hint equal to it can be dropped. */
|
|
438
|
+
export function defaultInlineDelims(n: TextNode): InlineDelimiter[] {
|
|
439
|
+
return defaultDelims(n)
|
|
440
|
+
}
|
|
441
|
+
|
|
442
|
+
const sameEntry = (a: Entry, b: Entry) => a.delim === b.delim && a.url === b.url
|
|
443
|
+
const opener = (e: Entry) => (e.mark !== "link" ? e.delim : e.delim === "link" ? "[" : e.delim === "<>" ? "<" : "")
|
|
444
|
+
const closer = (e: Entry) => (e.mark !== "link" ? e.delim : e.delim === "link" ? `](${e.url})` : e.delim === "<>" ? ">" : "")
|
|
445
|
+
|
|
446
|
+
function codeSpan(t: string): string {
|
|
447
|
+
const runs = new Set((t.match(/`+/g) ?? []).map((r) => r.length))
|
|
448
|
+
let n = 1
|
|
449
|
+
while (runs.has(n)) n++
|
|
450
|
+
const fence = "`".repeat(n)
|
|
451
|
+
const pad = t.startsWith("`") || t.endsWith("`") ? " " : ""
|
|
452
|
+
return `${fence}${pad}${t}${pad}${fence}`
|
|
453
|
+
}
|
|
454
|
+
|
|
455
|
+
/** A piece of output: markdown syntax (verbatim) or literal text (escaped on render). */
|
|
456
|
+
interface Piece {
|
|
457
|
+
s: string
|
|
458
|
+
text?: boolean
|
|
459
|
+
/** Literal text inside `[…]` link text, where `]` must be escaped. */
|
|
460
|
+
inLink?: boolean
|
|
461
|
+
}
|
|
462
|
+
|
|
463
|
+
/**
|
|
464
|
+
* Does literal `raw[i]` need a backslash? `minimal` escapes only what would otherwise parse as syntax —
|
|
465
|
+
* judged against the whole rendered line, so snake_case, `2 * 3`, `~5 min` and `a [b] c` stay as written.
|
|
466
|
+
*/
|
|
467
|
+
function needsEscape(raw: string, text: Uint8Array, inLink: Uint8Array, i: number, minimal: boolean) {
|
|
468
|
+
const c = raw[i]
|
|
469
|
+
switch (c) {
|
|
470
|
+
case "\\":
|
|
471
|
+
return !minimal || i + 1 >= raw.length || ASCII_PUNCT.test(raw[i + 1])
|
|
472
|
+
case "`":
|
|
473
|
+
return true
|
|
474
|
+
case "[":
|
|
475
|
+
return !minimal || raw[i + 1] === "^" || /^\[(?:\\.|[^\]\\])*\][([]/.test(raw.slice(i))
|
|
476
|
+
case "]":
|
|
477
|
+
return !minimal || inLink[i] === 1
|
|
478
|
+
case "@":
|
|
479
|
+
case "#":
|
|
480
|
+
case "$":
|
|
481
|
+
// Only at a chip boundary, so emails / C# / $5 survive untouched.
|
|
482
|
+
return isReferenceBoundary(raw[i - 1]) && /[A-Za-z_]/.test(raw[i + 1] ?? "")
|
|
483
|
+
case "*":
|
|
484
|
+
case "_":
|
|
485
|
+
case "~": {
|
|
486
|
+
if (!minimal) return true
|
|
487
|
+
let s = i
|
|
488
|
+
let e = i + 1
|
|
489
|
+
while (s > 0 && raw[s - 1] === c) s--
|
|
490
|
+
while (e < raw.length && raw[e] === c) e++
|
|
491
|
+
for (let k = s; k < e; k++) if (!text[k]) return true // touches a marker: always escape
|
|
492
|
+
if (c === "~") return e - s >= 2
|
|
493
|
+
const { open, close } = openClose(c, raw[s - 1], raw[e])
|
|
494
|
+
return open || close
|
|
495
|
+
}
|
|
496
|
+
}
|
|
497
|
+
return false
|
|
498
|
+
}
|
|
499
|
+
|
|
500
|
+
/** The rendered line: its characters, and which literal ones get a backslash. */
|
|
501
|
+
interface Rendering {
|
|
502
|
+
raw: string
|
|
503
|
+
escape: Uint8Array
|
|
504
|
+
}
|
|
505
|
+
|
|
506
|
+
function render(pieces: Piece[], minimal: boolean): Rendering {
|
|
507
|
+
const raw = pieces.map((p) => p.s).join("")
|
|
508
|
+
const text = new Uint8Array(raw.length)
|
|
509
|
+
const inLink = new Uint8Array(raw.length)
|
|
510
|
+
let at = 0
|
|
511
|
+
for (const p of pieces) {
|
|
512
|
+
if (p.text) text.fill(1, at, at + p.s.length)
|
|
513
|
+
if (p.inLink) inLink.fill(1, at, at + p.s.length)
|
|
514
|
+
at += p.s.length
|
|
515
|
+
}
|
|
516
|
+
const escape = new Uint8Array(raw.length)
|
|
517
|
+
for (let i = 0; i < raw.length; i++) {
|
|
518
|
+
if (text[i] && /[\\`*_~[\]@#$]/.test(raw[i]) && needsEscape(raw, text, inLink, i, minimal)) escape[i] = 1
|
|
519
|
+
}
|
|
520
|
+
return { raw, escape }
|
|
521
|
+
}
|
|
522
|
+
|
|
523
|
+
const written = ({ raw, escape }: Rendering) => {
|
|
245
524
|
let out = ""
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
525
|
+
for (let i = 0; i < raw.length; i++) out += escape[i] ? `\\${raw[i]}` : raw[i]
|
|
526
|
+
return out
|
|
527
|
+
}
|
|
528
|
+
|
|
529
|
+
/**
|
|
530
|
+
* Drop backslashes the line can do without — `footnote*`, a lone backtick, `[x]` before a `(` that isn't a
|
|
531
|
+
* link. An escaped run is dropped only if the line still means `want` with every escaped run of the same
|
|
532
|
+
* character unescaped too, so paired escapes (`\*x\*`, `` \`x\` ``) keep both backslashes.
|
|
533
|
+
*/
|
|
534
|
+
function relax(r: Rendering, want: string): string {
|
|
535
|
+
const original = written(r)
|
|
536
|
+
const groups: Array<{ start: number; end: number; ch: string }> = []
|
|
537
|
+
for (let i = 0; i < r.raw.length; i++) {
|
|
538
|
+
const ch = r.raw[i]
|
|
539
|
+
if (!r.escape[i] || !/[`*_~[]/.test(ch)) continue
|
|
540
|
+
let end = i + 1
|
|
541
|
+
if (ch !== "`" && ch !== "[") while (r.escape[end] && r.raw[end] === ch) end++
|
|
542
|
+
groups.push({ start: i, end, ch })
|
|
543
|
+
i = end - 1
|
|
544
|
+
}
|
|
545
|
+
if (!groups.length || groups.length > 32) return original
|
|
546
|
+
const without = (drop: (g: (typeof groups)[number]) => boolean) => {
|
|
547
|
+
const escape = r.escape.slice()
|
|
548
|
+
for (const g of groups) if (drop(g)) escape.fill(0, g.start, g.end)
|
|
549
|
+
return written({ raw: r.raw, escape })
|
|
249
550
|
}
|
|
551
|
+
const droppable = new Set(groups.filter((g) => meaning(parseInline(without((o) => o.ch === g.ch))) === want))
|
|
552
|
+
if (!droppable.size) return original
|
|
553
|
+
const relaxed = without((g) => droppable.has(g))
|
|
554
|
+
return meaning(parseInline(relaxed)) === want ? relaxed : original
|
|
555
|
+
}
|
|
556
|
+
|
|
557
|
+
/** What a line means, ignoring spelling: merged text runs with their marks, plus atoms. */
|
|
558
|
+
function meaning(nodes: Inline[]): string {
|
|
559
|
+
const parts: unknown[] = []
|
|
560
|
+
let last: { key: string; text: string } | undefined
|
|
250
561
|
for (const n of nodes) {
|
|
251
562
|
if (n.type !== "text") {
|
|
252
|
-
|
|
253
|
-
|
|
563
|
+
last = undefined
|
|
564
|
+
parts.push(n.type === "image" ? ["img", n.src, n.link ?? ""] : ["ref", n.kind, n.id])
|
|
254
565
|
continue
|
|
255
566
|
}
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
567
|
+
if (!n.text) continue
|
|
568
|
+
const key = JSON.stringify([!!n.bold, !!n.italic, !!n.strike, !!n.code, n.link ?? ""])
|
|
569
|
+
if (last && last.key === key && !n.code) last.text += n.text
|
|
570
|
+
else parts.push((last = { key, text: n.text }))
|
|
571
|
+
}
|
|
572
|
+
return JSON.stringify(parts)
|
|
573
|
+
}
|
|
574
|
+
|
|
575
|
+
// Serializes inline nodes back to markdown, as written: each run's `delims` restores the author's markers
|
|
576
|
+
// and nesting, and formatting shared by adjacent runs is opened once — `**bold [link](url) inside**` and
|
|
577
|
+
// `[**a** b](url)` come back whole instead of per run. Text is escaped minimally; if the minimal form would
|
|
578
|
+
// parse back differently, everything escapable is escaped instead.
|
|
579
|
+
export function serializeInline(nodes: Inline[]): string {
|
|
580
|
+
const pieces: Piece[] = []
|
|
581
|
+
const open: Entry[] = []
|
|
582
|
+
const closeTo = (depth: number) => {
|
|
583
|
+
while (open.length > depth) pieces.push({ s: closer(open.pop()!) })
|
|
584
|
+
}
|
|
585
|
+
for (const n of nodes) {
|
|
586
|
+
if (n.type !== "text") {
|
|
265
587
|
closeTo(0)
|
|
266
|
-
|
|
267
|
-
for (const m of [...wanted].reverse()) s = `${MARKER[m]}${s}${MARKER[m]}`
|
|
268
|
-
out += `[${s}](${n.link})`
|
|
588
|
+
pieces.push({ s: n.type === "image" ? serializeImage(n) : serializeReference(n) })
|
|
269
589
|
continue
|
|
270
590
|
}
|
|
271
|
-
|
|
591
|
+
if (!n.text) continue
|
|
592
|
+
const wanted = entriesFor(n)
|
|
593
|
+
// Keep the open formatting this run still wants (outermost first), close the rest, open what's missing.
|
|
594
|
+
const missing = [...wanted]
|
|
272
595
|
let keep = 0
|
|
273
|
-
|
|
596
|
+
for (; keep < open.length; keep++) {
|
|
597
|
+
const k = missing.findIndex((w) => sameEntry(w, open[keep]))
|
|
598
|
+
if (k < 0) break
|
|
599
|
+
missing.splice(k, 1)
|
|
600
|
+
}
|
|
274
601
|
closeTo(keep)
|
|
275
|
-
for (const
|
|
276
|
-
|
|
602
|
+
for (const w of missing) {
|
|
603
|
+
open.push(w)
|
|
604
|
+
pieces.push({ s: opener(w) })
|
|
605
|
+
}
|
|
606
|
+
const innermost = open[open.length - 1]
|
|
607
|
+
if (n.code) pieces.push({ s: codeSpan(n.text) })
|
|
608
|
+
else if (innermost?.mark === "link" && innermost.delim !== "link") pieces.push({ s: n.text }) // autolink
|
|
609
|
+
else pieces.push({ s: n.text, text: true, inLink: open.some((e) => e.delim === "link") })
|
|
277
610
|
}
|
|
278
611
|
closeTo(0)
|
|
279
|
-
|
|
612
|
+
|
|
613
|
+
const minimal = render(pieces, true)
|
|
614
|
+
if (!pieces.some((p) => p.text)) return written(minimal)
|
|
615
|
+
const want = meaning(nodes)
|
|
616
|
+
if (meaning(parseInline(written(minimal))) === want) return relax(minimal, want)
|
|
617
|
+
const full = written(render(pieces, false))
|
|
618
|
+
return meaning(parseInline(full)) === want ? full : written(minimal)
|
|
280
619
|
}
|
|
281
620
|
|
|
282
621
|
export function plainText(nodes: Inline[]): string {
|