pantsdown 1.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.eslintrc.json +26 -0
- package/.github/workflows/release.yaml +44 -0
- package/.prettierrc.json +11 -0
- package/LICENSE.md +67 -0
- package/README.md +5 -0
- package/bun.lockb +0 -0
- package/package.json +46 -0
- package/src/index.ts +2 -0
- package/src/lexer.ts +369 -0
- package/src/pantsdown.ts +19 -0
- package/src/parser.ts +216 -0
- package/src/renderer.ts +145 -0
- package/src/rules/block.ts +129 -0
- package/src/rules/inline.ts +146 -0
- package/src/rules/utils.ts +18 -0
- package/src/sample.test.ts +5 -0
- package/src/tokenizer.ts +773 -0
- package/src/types.ts +183 -0
- package/src/utils.ts +231 -0
- package/tsconfig.json +33 -0
package/src/tokenizer.ts
ADDED
|
@@ -0,0 +1,773 @@
|
|
|
1
|
+
import type { Lexer } from "./lexer.ts";
|
|
2
|
+
import { block } from "./rules/block.ts";
|
|
3
|
+
import { inline } from "./rules/inline.ts";
|
|
4
|
+
import { type Links, type Tokens } from "./types.ts";
|
|
5
|
+
import {
|
|
6
|
+
escape,
|
|
7
|
+
findClosingBracket,
|
|
8
|
+
indentCodeCompensation,
|
|
9
|
+
outputLink,
|
|
10
|
+
rtrim,
|
|
11
|
+
splitCells,
|
|
12
|
+
} from "./utils.ts";
|
|
13
|
+
|
|
14
|
+
/**
|
|
15
|
+
* The tokenizer defines how to turn markdown text into tokens.
|
|
16
|
+
*/
|
|
17
|
+
export class Tokenizer {
|
|
18
|
+
private lexer: Lexer;
|
|
19
|
+
|
|
20
|
+
constructor(lexer: Lexer) {
|
|
21
|
+
this.lexer = lexer;
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
space(src: string): Tokens["Space"] | undefined {
|
|
25
|
+
const cap = block.newline.exec(src);
|
|
26
|
+
if (cap && cap[0].length > 0) {
|
|
27
|
+
return {
|
|
28
|
+
type: "space",
|
|
29
|
+
raw: cap[0],
|
|
30
|
+
};
|
|
31
|
+
}
|
|
32
|
+
return;
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
code(src: string): Tokens["Code"] | undefined {
|
|
36
|
+
const cap = block.code.exec(src);
|
|
37
|
+
if (!cap) return undefined;
|
|
38
|
+
|
|
39
|
+
const text = cap[0].replace(/^ {1,4}/gm, "");
|
|
40
|
+
return {
|
|
41
|
+
type: "code",
|
|
42
|
+
raw: cap[0],
|
|
43
|
+
codeBlockStyle: "indented",
|
|
44
|
+
text: rtrim(text, "\n"),
|
|
45
|
+
};
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
fences(src: string): Tokens["Code"] | undefined {
|
|
49
|
+
const cap = block.fences.exec(src);
|
|
50
|
+
if (!cap) return undefined;
|
|
51
|
+
|
|
52
|
+
const raw = cap[0];
|
|
53
|
+
const text = indentCodeCompensation(raw, cap[3] ?? "");
|
|
54
|
+
|
|
55
|
+
return {
|
|
56
|
+
type: "code",
|
|
57
|
+
raw,
|
|
58
|
+
lang: cap[2] ? cap[2].trim().replace(inline.escapes, "$1") : cap[2],
|
|
59
|
+
text,
|
|
60
|
+
};
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
heading(src: string): Tokens["Heading"] | undefined {
|
|
64
|
+
const cap = block.heading.exec(src);
|
|
65
|
+
if (!cap) return undefined;
|
|
66
|
+
let text = cap[2]!.trim();
|
|
67
|
+
|
|
68
|
+
// remove trailing #s
|
|
69
|
+
if (text.endsWith("#")) {
|
|
70
|
+
const trimmed = rtrim(text, "#");
|
|
71
|
+
if (!trimmed || trimmed.endsWith(" ")) {
|
|
72
|
+
// CommonMark requires space before trailing #s
|
|
73
|
+
text = trimmed.trim();
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
return {
|
|
78
|
+
type: "heading",
|
|
79
|
+
raw: cap[0],
|
|
80
|
+
depth: cap[1]!.length,
|
|
81
|
+
text,
|
|
82
|
+
tokens: this.lexer.inline(text),
|
|
83
|
+
};
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
hr(src: string): Tokens["Hr"] | undefined {
|
|
87
|
+
const cap = block.hr.exec(src);
|
|
88
|
+
if (!cap) return undefined;
|
|
89
|
+
|
|
90
|
+
return {
|
|
91
|
+
type: "hr",
|
|
92
|
+
raw: cap[0],
|
|
93
|
+
};
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
blockquote(src: string): Tokens["Blockquote"] | undefined {
|
|
97
|
+
const cap = block.blockquote.exec(src);
|
|
98
|
+
if (!cap) return undefined;
|
|
99
|
+
|
|
100
|
+
const text = cap[0].replace(/^ *>[ \t]?/gm, "");
|
|
101
|
+
const top = this.lexer.state.top;
|
|
102
|
+
this.lexer.state.top = true;
|
|
103
|
+
const tokens = this.lexer.blockTokens(text, []);
|
|
104
|
+
this.lexer.state.top = top;
|
|
105
|
+
return {
|
|
106
|
+
type: "blockquote",
|
|
107
|
+
raw: cap[0],
|
|
108
|
+
tokens,
|
|
109
|
+
text,
|
|
110
|
+
};
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
list(src: string): Tokens["List"] | undefined {
|
|
114
|
+
let cap = block.list.exec(src);
|
|
115
|
+
if (!cap) return undefined;
|
|
116
|
+
|
|
117
|
+
let bull = cap[1]!.trim();
|
|
118
|
+
const isordered = bull.length > 1;
|
|
119
|
+
|
|
120
|
+
const list: Tokens["List"] = {
|
|
121
|
+
type: "list",
|
|
122
|
+
raw: "",
|
|
123
|
+
ordered: isordered,
|
|
124
|
+
start: isordered ? +bull.slice(0, -1) : "",
|
|
125
|
+
loose: false,
|
|
126
|
+
items: [] as Tokens["ListItem"][],
|
|
127
|
+
};
|
|
128
|
+
|
|
129
|
+
bull = isordered ? `\\d{1,9}\\${bull.slice(-1)}` : `\\${bull}`;
|
|
130
|
+
|
|
131
|
+
// Get next list item
|
|
132
|
+
const itemRegex = new RegExp(`^( {0,3}${bull})((?:[\t ][^\\n]*)?(?:\\n|$))`);
|
|
133
|
+
let raw = "";
|
|
134
|
+
let itemContents = "";
|
|
135
|
+
let endsWithBlankLine = false;
|
|
136
|
+
// Check if current bullet point can start a new List Item
|
|
137
|
+
while (src) {
|
|
138
|
+
let endEarly = false;
|
|
139
|
+
if (!(cap = itemRegex.exec(src))) {
|
|
140
|
+
break;
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
if (block.hr.test(src)) {
|
|
144
|
+
// End list if bullet was actually HR (possibly move into itemRegex?)
|
|
145
|
+
break;
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
raw = cap[0];
|
|
149
|
+
src = src.substring(raw.length);
|
|
150
|
+
|
|
151
|
+
let line = cap[2]!
|
|
152
|
+
.split("\n", 1)[0]!
|
|
153
|
+
.replace(/^\t+/, (t: string) => " ".repeat(3 * t.length));
|
|
154
|
+
let nextLine = src.split("\n", 1)[0] ?? "";
|
|
155
|
+
|
|
156
|
+
let indent = 0;
|
|
157
|
+
indent = cap[2]!.search(/[^ ]/); // Find first non-space char
|
|
158
|
+
indent = indent > 4 ? 1 : indent; // Treat indented code blocks (> 4 spaces) as having only 1 indent
|
|
159
|
+
itemContents = line.slice(indent);
|
|
160
|
+
indent += cap[1]!.length;
|
|
161
|
+
|
|
162
|
+
let blankLine = false;
|
|
163
|
+
|
|
164
|
+
if (!line && /^ *$/.test(nextLine)) {
|
|
165
|
+
// Items begin with at most one blank line
|
|
166
|
+
raw += nextLine + "\n";
|
|
167
|
+
src = src.substring(nextLine.length + 1);
|
|
168
|
+
endEarly = true;
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
if (!endEarly) {
|
|
172
|
+
const nextBulletRegex = new RegExp(
|
|
173
|
+
`^ {0,${Math.min(
|
|
174
|
+
3,
|
|
175
|
+
indent - 1,
|
|
176
|
+
)}}(?:[*+-]|\\d{1,9}[.)])((?:[ \t][^\\n]*)?(?:\\n|$))`,
|
|
177
|
+
);
|
|
178
|
+
const hrRegex = new RegExp(
|
|
179
|
+
`^ {0,${Math.min(
|
|
180
|
+
3,
|
|
181
|
+
indent - 1,
|
|
182
|
+
)}}((?:- *){3,}|(?:_ *){3,}|(?:\\* *){3,})(?:\\n+|$)`,
|
|
183
|
+
);
|
|
184
|
+
const fencesBeginRegex = new RegExp(
|
|
185
|
+
`^ {0,${Math.min(3, indent - 1)}}(?:\`\`\`|~~~)`,
|
|
186
|
+
);
|
|
187
|
+
const headingBeginRegex = new RegExp(`^ {0,${Math.min(3, indent - 1)}}#`);
|
|
188
|
+
|
|
189
|
+
// Check if following lines should be included in List Item
|
|
190
|
+
while (src) {
|
|
191
|
+
const rawLine = src.split("\n", 1)[0] ?? "";
|
|
192
|
+
nextLine = rawLine;
|
|
193
|
+
|
|
194
|
+
// End list item if found code fences
|
|
195
|
+
if (fencesBeginRegex.test(nextLine)) {
|
|
196
|
+
break;
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
// End list item if found start of new heading
|
|
200
|
+
if (headingBeginRegex.test(nextLine)) {
|
|
201
|
+
break;
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
// End list item if found start of new bullet
|
|
205
|
+
if (nextBulletRegex.test(nextLine)) {
|
|
206
|
+
break;
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
// Horizontal rule found
|
|
210
|
+
if (hrRegex.test(src)) {
|
|
211
|
+
break;
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
if (nextLine.search(/[^ ]/) >= indent || !nextLine.trim()) {
|
|
215
|
+
// Dedent if possible
|
|
216
|
+
itemContents += "\n" + nextLine.slice(indent);
|
|
217
|
+
} else {
|
|
218
|
+
// not enough indentation
|
|
219
|
+
if (blankLine) {
|
|
220
|
+
break;
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
// paragraph continuation unless last line was a different block level element
|
|
224
|
+
if (line.search(/[^ ]/) >= 4) {
|
|
225
|
+
// indented code block
|
|
226
|
+
break;
|
|
227
|
+
}
|
|
228
|
+
if (fencesBeginRegex.test(line)) {
|
|
229
|
+
break;
|
|
230
|
+
}
|
|
231
|
+
if (headingBeginRegex.test(line)) {
|
|
232
|
+
break;
|
|
233
|
+
}
|
|
234
|
+
if (hrRegex.test(line)) {
|
|
235
|
+
break;
|
|
236
|
+
}
|
|
237
|
+
|
|
238
|
+
itemContents += "\n" + nextLine;
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
if (!blankLine && !nextLine.trim()) {
|
|
242
|
+
// Check if current line is blank
|
|
243
|
+
blankLine = true;
|
|
244
|
+
}
|
|
245
|
+
|
|
246
|
+
raw += rawLine + "\n";
|
|
247
|
+
src = src.substring(rawLine.length + 1);
|
|
248
|
+
line = nextLine.slice(indent);
|
|
249
|
+
}
|
|
250
|
+
}
|
|
251
|
+
|
|
252
|
+
if (!list.loose) {
|
|
253
|
+
// If the previous item ended with a blank line, the list is loose
|
|
254
|
+
if (endsWithBlankLine) {
|
|
255
|
+
list.loose = true;
|
|
256
|
+
} else if (/\n *\n *$/.test(raw)) {
|
|
257
|
+
endsWithBlankLine = true;
|
|
258
|
+
}
|
|
259
|
+
}
|
|
260
|
+
|
|
261
|
+
let istask: RegExpExecArray | null = null;
|
|
262
|
+
let ischecked: boolean | undefined;
|
|
263
|
+
// Check for task list items
|
|
264
|
+
istask = /^\[[ xX]\] /.exec(itemContents);
|
|
265
|
+
if (istask) {
|
|
266
|
+
ischecked = istask[0] !== "[ ] ";
|
|
267
|
+
itemContents = itemContents.replace(/^\[[ xX]\] +/, "");
|
|
268
|
+
}
|
|
269
|
+
|
|
270
|
+
list.items.push({
|
|
271
|
+
type: "list_item",
|
|
272
|
+
raw,
|
|
273
|
+
task: Boolean(istask),
|
|
274
|
+
checked: ischecked,
|
|
275
|
+
loose: false,
|
|
276
|
+
text: itemContents,
|
|
277
|
+
tokens: [],
|
|
278
|
+
});
|
|
279
|
+
|
|
280
|
+
list.raw += raw;
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
// Do not consume newlines at end of final item. Alternatively, make itemRegex *start* with any newlines to simplify/speed up endsWithBlankLine logic
|
|
284
|
+
list.items[list.items.length - 1]!.raw = raw.trimEnd();
|
|
285
|
+
list.items[list.items.length - 1]!.text = itemContents.trimEnd();
|
|
286
|
+
list.raw = list.raw.trimEnd();
|
|
287
|
+
|
|
288
|
+
// Item child tokens handled here at end because we needed to have the final item to trim it first
|
|
289
|
+
for (let i = 0, listItemsLen = list.items.length; i < listItemsLen; i++) {
|
|
290
|
+
this.lexer.state.top = false;
|
|
291
|
+
list.items[i]!.tokens = this.lexer.blockTokens(list.items[i]!.text, []);
|
|
292
|
+
|
|
293
|
+
if (!list.loose) {
|
|
294
|
+
// Check if list should be loose
|
|
295
|
+
const spacers = list.items[i]!.tokens.filter((t) => t.type === "space");
|
|
296
|
+
const hasMultipleLineBreaks =
|
|
297
|
+
// eslint-disable-next-line
|
|
298
|
+
spacers.length > 0 && spacers.some((t: any) => /\n.*\n/.test(t.raw));
|
|
299
|
+
|
|
300
|
+
list.loose = hasMultipleLineBreaks;
|
|
301
|
+
}
|
|
302
|
+
}
|
|
303
|
+
|
|
304
|
+
// Set all items to loose if list is loose
|
|
305
|
+
if (list.loose) {
|
|
306
|
+
for (let i = 0, listItemsLen = list.items.length; i < listItemsLen; i++) {
|
|
307
|
+
list.items[i]!.loose = true;
|
|
308
|
+
}
|
|
309
|
+
}
|
|
310
|
+
|
|
311
|
+
return list;
|
|
312
|
+
}
|
|
313
|
+
|
|
314
|
+
html(src: string): Tokens["HTML"] | undefined {
|
|
315
|
+
const cap = block.html.exec(src);
|
|
316
|
+
if (!cap) return undefined;
|
|
317
|
+
|
|
318
|
+
const token: Tokens["HTML"] = {
|
|
319
|
+
type: "html",
|
|
320
|
+
block: true,
|
|
321
|
+
raw: cap[0],
|
|
322
|
+
pre: cap[1] === "pre" || cap[1] === "script" || cap[1] === "style",
|
|
323
|
+
text: cap[0],
|
|
324
|
+
};
|
|
325
|
+
return token;
|
|
326
|
+
}
|
|
327
|
+
|
|
328
|
+
def(src: string): Tokens["Def"] | undefined {
|
|
329
|
+
const cap = block.def.exec(src);
|
|
330
|
+
if (!cap) return undefined;
|
|
331
|
+
|
|
332
|
+
const tag = cap[1]!.toLowerCase().replace(/\s+/g, " ");
|
|
333
|
+
const href = cap[2] ? cap[2].replace(/^<(.*)>$/, "$1").replace(inline.escapes, "$1") : "";
|
|
334
|
+
const title = cap[3]
|
|
335
|
+
? cap[3].substring(1, cap[3].length - 1).replace(inline.escapes, "$1")
|
|
336
|
+
: "";
|
|
337
|
+
return {
|
|
338
|
+
type: "def",
|
|
339
|
+
tag,
|
|
340
|
+
raw: cap[0],
|
|
341
|
+
href,
|
|
342
|
+
title,
|
|
343
|
+
};
|
|
344
|
+
}
|
|
345
|
+
|
|
346
|
+
table(src: string): Tokens["Table"] | undefined {
|
|
347
|
+
const cap = block.table.exec(src);
|
|
348
|
+
if (!cap?.[2]) return;
|
|
349
|
+
|
|
350
|
+
if (!/[:|]/.test(cap[2])) {
|
|
351
|
+
// delimiter row must have a pipe (|) or colon (:) otherwise it is a setext heading
|
|
352
|
+
return;
|
|
353
|
+
}
|
|
354
|
+
|
|
355
|
+
const item: Tokens["Table"] = {
|
|
356
|
+
type: "table",
|
|
357
|
+
raw: cap[0],
|
|
358
|
+
header: splitCells(cap[1]!).map((c) => ({
|
|
359
|
+
type: "tablecell",
|
|
360
|
+
raw: c,
|
|
361
|
+
text: c,
|
|
362
|
+
tokens: [],
|
|
363
|
+
})),
|
|
364
|
+
align: [],
|
|
365
|
+
rows: [],
|
|
366
|
+
};
|
|
367
|
+
|
|
368
|
+
const align = cap[2].replace(/^\||\| *$/g, "").split("|") as (string | null)[];
|
|
369
|
+
const rows = cap[3]?.trim() ? cap[3].replace(/\n[ \t]*$/, "").split("\n") : [];
|
|
370
|
+
|
|
371
|
+
if (item.header.length !== align.length) return;
|
|
372
|
+
|
|
373
|
+
let l = align.length;
|
|
374
|
+
let i, j, k, row;
|
|
375
|
+
for (i = 0; i < l; i++) {
|
|
376
|
+
const alignStr = align[i];
|
|
377
|
+
if (alignStr) {
|
|
378
|
+
if (/^ *-+: *$/.test(alignStr)) {
|
|
379
|
+
item.align.push("right");
|
|
380
|
+
} else if (/^ *:-+: *$/.test(alignStr)) {
|
|
381
|
+
item.align.push("center");
|
|
382
|
+
} else if (/^ *:-+ *$/.test(alignStr)) {
|
|
383
|
+
item.align.push("left");
|
|
384
|
+
} else {
|
|
385
|
+
item.align.push(null);
|
|
386
|
+
}
|
|
387
|
+
}
|
|
388
|
+
}
|
|
389
|
+
|
|
390
|
+
l = rows.length;
|
|
391
|
+
for (i = 0; i < l; i++) {
|
|
392
|
+
item.rows.push(
|
|
393
|
+
splitCells(rows[i] as unknown as string, item.header.length).map((c) => ({
|
|
394
|
+
type: "tablecell",
|
|
395
|
+
raw: c,
|
|
396
|
+
text: c,
|
|
397
|
+
tokens: [],
|
|
398
|
+
})),
|
|
399
|
+
);
|
|
400
|
+
}
|
|
401
|
+
|
|
402
|
+
// parse child tokens inside headers and cells
|
|
403
|
+
|
|
404
|
+
// header child tokens
|
|
405
|
+
l = item.header.length;
|
|
406
|
+
for (j = 0; j < l; j++) {
|
|
407
|
+
item.header[j]!.tokens = this.lexer.inline(item.header[j]!.text);
|
|
408
|
+
}
|
|
409
|
+
|
|
410
|
+
// cell child tokens
|
|
411
|
+
l = item.rows.length;
|
|
412
|
+
for (j = 0; j < l; j++) {
|
|
413
|
+
row = item.rows[j]!;
|
|
414
|
+
for (k = 0; k < row.length; k++) {
|
|
415
|
+
row[k]!.tokens = this.lexer.inline(row[k]!.text);
|
|
416
|
+
}
|
|
417
|
+
}
|
|
418
|
+
|
|
419
|
+
return item;
|
|
420
|
+
}
|
|
421
|
+
|
|
422
|
+
lheading(src: string): Tokens["Heading"] | undefined {
|
|
423
|
+
const cap = block.lheading.exec(src);
|
|
424
|
+
if (!cap) return undefined;
|
|
425
|
+
|
|
426
|
+
return {
|
|
427
|
+
type: "heading",
|
|
428
|
+
raw: cap[0],
|
|
429
|
+
depth: cap[2]!.startsWith("=") ? 1 : 2,
|
|
430
|
+
text: cap[1]!,
|
|
431
|
+
tokens: this.lexer.inline(cap[1]!),
|
|
432
|
+
};
|
|
433
|
+
}
|
|
434
|
+
|
|
435
|
+
paragraph(src: string): Tokens["Paragraph"] | undefined {
|
|
436
|
+
const cap = block.paragraph.exec(src);
|
|
437
|
+
if (!cap) return undefined;
|
|
438
|
+
|
|
439
|
+
const text = cap[1]!.endsWith("\n") ? cap[1]!.slice(0, -1) : cap[1]!;
|
|
440
|
+
return {
|
|
441
|
+
type: "paragraph",
|
|
442
|
+
raw: cap[0],
|
|
443
|
+
text,
|
|
444
|
+
tokens: this.lexer.inline(text),
|
|
445
|
+
};
|
|
446
|
+
}
|
|
447
|
+
|
|
448
|
+
text(src: string): Tokens["Text"] | undefined {
|
|
449
|
+
const cap = block.text.exec(src);
|
|
450
|
+
if (!cap) return undefined;
|
|
451
|
+
|
|
452
|
+
return {
|
|
453
|
+
type: "text",
|
|
454
|
+
raw: cap[0],
|
|
455
|
+
text: cap[0],
|
|
456
|
+
tokens: this.lexer.inline(cap[0]),
|
|
457
|
+
};
|
|
458
|
+
}
|
|
459
|
+
|
|
460
|
+
escape(src: string): Tokens["Escape"] | undefined {
|
|
461
|
+
const cap = inline.escape.exec(src);
|
|
462
|
+
if (!cap) return undefined;
|
|
463
|
+
|
|
464
|
+
return {
|
|
465
|
+
type: "escape",
|
|
466
|
+
raw: cap[0],
|
|
467
|
+
text: escape(cap[1]!),
|
|
468
|
+
};
|
|
469
|
+
}
|
|
470
|
+
|
|
471
|
+
tag(src: string): Tokens["Tag"] | undefined {
|
|
472
|
+
const cap = inline.tag.exec(src);
|
|
473
|
+
if (!cap) return undefined;
|
|
474
|
+
|
|
475
|
+
if (!this.lexer.state.inLink && /^<a /i.test(cap[0])) {
|
|
476
|
+
this.lexer.state.inLink = true;
|
|
477
|
+
} else if (this.lexer.state.inLink && /^<\/a>/i.test(cap[0])) {
|
|
478
|
+
this.lexer.state.inLink = false;
|
|
479
|
+
}
|
|
480
|
+
if (!this.lexer.state.inRawBlock && /^<(pre|code|kbd|script)(\s|>)/i.test(cap[0])) {
|
|
481
|
+
this.lexer.state.inRawBlock = true;
|
|
482
|
+
} else if (this.lexer.state.inRawBlock && /^<\/(pre|code|kbd|script)(\s|>)/i.test(cap[0])) {
|
|
483
|
+
this.lexer.state.inRawBlock = false;
|
|
484
|
+
}
|
|
485
|
+
|
|
486
|
+
return {
|
|
487
|
+
type: "html",
|
|
488
|
+
raw: cap[0],
|
|
489
|
+
inLink: this.lexer.state.inLink,
|
|
490
|
+
inRawBlock: this.lexer.state.inRawBlock,
|
|
491
|
+
block: false,
|
|
492
|
+
text: cap[0],
|
|
493
|
+
};
|
|
494
|
+
}
|
|
495
|
+
|
|
496
|
+
link(src: string): Tokens["Link"] | Tokens["Image"] | undefined {
|
|
497
|
+
const cap = inline.link.exec(src);
|
|
498
|
+
if (!cap) return undefined;
|
|
499
|
+
|
|
500
|
+
const trimmedUrl = cap[2]!.trim();
|
|
501
|
+
if (trimmedUrl.startsWith("<")) {
|
|
502
|
+
// commonmark requires matching angle brackets
|
|
503
|
+
if (!trimmedUrl.endsWith(">")) {
|
|
504
|
+
return;
|
|
505
|
+
}
|
|
506
|
+
|
|
507
|
+
// ending angle bracket cannot be escaped
|
|
508
|
+
const rtrimSlash = rtrim(trimmedUrl.slice(0, -1), "\\");
|
|
509
|
+
if ((trimmedUrl.length - rtrimSlash.length) % 2 === 0) {
|
|
510
|
+
return;
|
|
511
|
+
}
|
|
512
|
+
} else {
|
|
513
|
+
// find closing parenthesis
|
|
514
|
+
const lastParenIndex = findClosingBracket(cap[2]!, "()");
|
|
515
|
+
if (lastParenIndex > -1) {
|
|
516
|
+
const start = cap[0].startsWith("!") ? 5 : 4;
|
|
517
|
+
const linkLen = start + cap[1]!.length + lastParenIndex;
|
|
518
|
+
cap[2] = cap[2]!.substring(0, lastParenIndex);
|
|
519
|
+
cap[0] = cap[0].substring(0, linkLen).trim();
|
|
520
|
+
cap[3] = "";
|
|
521
|
+
}
|
|
522
|
+
}
|
|
523
|
+
let href = cap[2]!;
|
|
524
|
+
let title = "";
|
|
525
|
+
title = cap[3] ? cap[3].slice(1, -1) : "";
|
|
526
|
+
|
|
527
|
+
href = href.trim();
|
|
528
|
+
if (href.startsWith("<")) {
|
|
529
|
+
href = href.slice(1, -1);
|
|
530
|
+
}
|
|
531
|
+
return outputLink(
|
|
532
|
+
cap,
|
|
533
|
+
{
|
|
534
|
+
href: href ? href.replace(inline.escapes, "$1") : href,
|
|
535
|
+
title: title ? title.replace(inline.escapes, "$1") : title,
|
|
536
|
+
},
|
|
537
|
+
cap[0],
|
|
538
|
+
this.lexer,
|
|
539
|
+
);
|
|
540
|
+
}
|
|
541
|
+
|
|
542
|
+
reflink(
|
|
543
|
+
src: string,
|
|
544
|
+
links: Links,
|
|
545
|
+
): Tokens["Link"] | Tokens["Image"] | Tokens["Text"] | undefined {
|
|
546
|
+
let cap;
|
|
547
|
+
if ((cap = inline.reflink.exec(src)) ?? (cap = inline.nolink.exec(src))) {
|
|
548
|
+
const linkStr = (cap[2] ?? cap[1])!.replace(/\s+/g, " ");
|
|
549
|
+
const link = links[linkStr.toLowerCase()];
|
|
550
|
+
if (!link) {
|
|
551
|
+
const text = cap[0].charAt(0);
|
|
552
|
+
return {
|
|
553
|
+
type: "text",
|
|
554
|
+
raw: text,
|
|
555
|
+
text,
|
|
556
|
+
};
|
|
557
|
+
}
|
|
558
|
+
return outputLink(cap, link, cap[0], this.lexer);
|
|
559
|
+
}
|
|
560
|
+
return undefined;
|
|
561
|
+
}
|
|
562
|
+
|
|
563
|
+
emStrong(
|
|
564
|
+
src: string,
|
|
565
|
+
maskedSrc: string,
|
|
566
|
+
prevChar = "",
|
|
567
|
+
): Tokens["Em"] | Tokens["Strong"] | undefined {
|
|
568
|
+
let match = inline.emStrong.lDelim.exec(src);
|
|
569
|
+
if (!match) return;
|
|
570
|
+
|
|
571
|
+
// _ can't be between two alphanumerics. \p{L}\p{N} includes non-english alphabet/numbers as well
|
|
572
|
+
if (match[3] && prevChar.match(/[\p{L}\p{N}]/u)) return;
|
|
573
|
+
|
|
574
|
+
// eslint-disable-next-line
|
|
575
|
+
const nextChar = match[1] || match[2] || "";
|
|
576
|
+
|
|
577
|
+
if (!nextChar || !prevChar || inline.punctuation.exec(prevChar)) {
|
|
578
|
+
// unicode Regex counts emoji as 1 char; spread into array for proper count (used multiple times below)
|
|
579
|
+
const lLength = [...match[0]].length - 1;
|
|
580
|
+
let rDelim,
|
|
581
|
+
rLength,
|
|
582
|
+
delimTotal = lLength,
|
|
583
|
+
midDelimTotal = 0;
|
|
584
|
+
|
|
585
|
+
const endReg = match[0].startsWith("*")
|
|
586
|
+
? inline.emStrong.rDelimAst
|
|
587
|
+
: inline.emStrong.rDelimUnd;
|
|
588
|
+
endReg.lastIndex = 0;
|
|
589
|
+
|
|
590
|
+
// Clip maskedSrc to same section of string as src (move to lexer?)
|
|
591
|
+
maskedSrc = maskedSrc.slice(-1 * src.length + match[0].length - 1);
|
|
592
|
+
|
|
593
|
+
while ((match = endReg.exec(maskedSrc)) != null) {
|
|
594
|
+
// eslint-disable-next-line
|
|
595
|
+
rDelim = match[1] || match[2] || match[3] || match[4] || match[5] || match[6];
|
|
596
|
+
|
|
597
|
+
if (!rDelim) continue; // skip single * in __abc*abc__
|
|
598
|
+
|
|
599
|
+
rLength = [...rDelim].length;
|
|
600
|
+
|
|
601
|
+
// eslint-disable-next-line
|
|
602
|
+
if (match[3] || match[4]) {
|
|
603
|
+
// found another Left Delim
|
|
604
|
+
delimTotal += rLength;
|
|
605
|
+
continue;
|
|
606
|
+
// eslint-disable-next-line
|
|
607
|
+
} else if (match[5] || match[6]) {
|
|
608
|
+
// either Left or Right Delim
|
|
609
|
+
if (lLength % 3 && !((lLength + rLength) % 3)) {
|
|
610
|
+
midDelimTotal += rLength;
|
|
611
|
+
continue; // CommonMark Emphasis Rules 9-10
|
|
612
|
+
}
|
|
613
|
+
}
|
|
614
|
+
|
|
615
|
+
delimTotal -= rLength;
|
|
616
|
+
|
|
617
|
+
if (delimTotal > 0) continue; // Haven't found enough closing delimiters
|
|
618
|
+
|
|
619
|
+
// Remove extra characters. *a*** -> *a*
|
|
620
|
+
rLength = Math.min(rLength, rLength + delimTotal + midDelimTotal);
|
|
621
|
+
|
|
622
|
+
const raw = [...src].slice(0, lLength + match.index + rLength + 1).join("");
|
|
623
|
+
|
|
624
|
+
// Create `em` if smallest delimiter has odd char count. *a***
|
|
625
|
+
if (Math.min(lLength, rLength) % 2) {
|
|
626
|
+
const text = raw.slice(1, -1);
|
|
627
|
+
return {
|
|
628
|
+
type: "em",
|
|
629
|
+
raw,
|
|
630
|
+
text,
|
|
631
|
+
tokens: this.lexer.inlineTokens(text),
|
|
632
|
+
};
|
|
633
|
+
}
|
|
634
|
+
|
|
635
|
+
// Create 'strong' if smallest delimiter has even char count. **a***
|
|
636
|
+
const text = raw.slice(2, -2);
|
|
637
|
+
return {
|
|
638
|
+
type: "strong",
|
|
639
|
+
raw,
|
|
640
|
+
text,
|
|
641
|
+
tokens: this.lexer.inlineTokens(text),
|
|
642
|
+
};
|
|
643
|
+
}
|
|
644
|
+
}
|
|
645
|
+
|
|
646
|
+
return undefined;
|
|
647
|
+
}
|
|
648
|
+
|
|
649
|
+
codespan(src: string): Tokens["Codespan"] | undefined {
|
|
650
|
+
const cap = inline.code.exec(src);
|
|
651
|
+
if (!cap) return undefined;
|
|
652
|
+
|
|
653
|
+
let text = cap[2]!.replace(/\n/g, " ");
|
|
654
|
+
const hasNonSpaceChars = /[^ ]/.test(text);
|
|
655
|
+
const hasSpaceCharsOnBothEnds = text.startsWith(" ") && text.endsWith(" ");
|
|
656
|
+
if (hasNonSpaceChars && hasSpaceCharsOnBothEnds) {
|
|
657
|
+
text = text.substring(1, text.length - 1);
|
|
658
|
+
}
|
|
659
|
+
text = escape(text, true);
|
|
660
|
+
return {
|
|
661
|
+
type: "codespan",
|
|
662
|
+
raw: cap[0],
|
|
663
|
+
text,
|
|
664
|
+
};
|
|
665
|
+
}
|
|
666
|
+
|
|
667
|
+
br(src: string): Tokens["Br"] | undefined {
|
|
668
|
+
const cap = inline.br.exec(src);
|
|
669
|
+
if (!cap) return undefined;
|
|
670
|
+
|
|
671
|
+
return {
|
|
672
|
+
type: "br",
|
|
673
|
+
raw: cap[0],
|
|
674
|
+
};
|
|
675
|
+
}
|
|
676
|
+
|
|
677
|
+
del(src: string): Tokens["Del"] | undefined {
|
|
678
|
+
const cap = inline.del.exec(src);
|
|
679
|
+
if (!cap) return undefined;
|
|
680
|
+
|
|
681
|
+
return {
|
|
682
|
+
type: "del",
|
|
683
|
+
raw: cap[0],
|
|
684
|
+
text: cap[2]!,
|
|
685
|
+
tokens: this.lexer.inlineTokens(cap[2]!),
|
|
686
|
+
};
|
|
687
|
+
}
|
|
688
|
+
|
|
689
|
+
autolink(src: string): Tokens["Link"] | undefined {
|
|
690
|
+
const cap = inline.autolink.exec(src);
|
|
691
|
+
if (!cap) return undefined;
|
|
692
|
+
|
|
693
|
+
let text, href;
|
|
694
|
+
if (cap[2] === "@") {
|
|
695
|
+
text = escape(cap[1]!);
|
|
696
|
+
href = "mailto:" + text;
|
|
697
|
+
} else {
|
|
698
|
+
text = escape(cap[1]!);
|
|
699
|
+
href = text;
|
|
700
|
+
}
|
|
701
|
+
|
|
702
|
+
return {
|
|
703
|
+
type: "link",
|
|
704
|
+
title: null,
|
|
705
|
+
raw: cap[0],
|
|
706
|
+
text,
|
|
707
|
+
href,
|
|
708
|
+
tokens: [
|
|
709
|
+
{
|
|
710
|
+
type: "text",
|
|
711
|
+
raw: text,
|
|
712
|
+
text,
|
|
713
|
+
},
|
|
714
|
+
],
|
|
715
|
+
};
|
|
716
|
+
}
|
|
717
|
+
|
|
718
|
+
url(src: string): Tokens["Link"] | undefined {
|
|
719
|
+
let cap;
|
|
720
|
+
if ((cap = inline.url.exec(src))) {
|
|
721
|
+
let text, href;
|
|
722
|
+
if (cap[2] === "@") {
|
|
723
|
+
text = escape(cap[0]);
|
|
724
|
+
href = "mailto:" + text;
|
|
725
|
+
} else {
|
|
726
|
+
// do extended autolink path validation
|
|
727
|
+
let prevCapZero;
|
|
728
|
+
do {
|
|
729
|
+
prevCapZero = cap[0];
|
|
730
|
+
cap[0] = inline.backpedal.exec(cap[0])![0];
|
|
731
|
+
} while (prevCapZero !== cap[0]);
|
|
732
|
+
text = escape(cap[0]);
|
|
733
|
+
if (cap[1] === "www.") {
|
|
734
|
+
href = "http://" + cap[0];
|
|
735
|
+
} else {
|
|
736
|
+
href = cap[0];
|
|
737
|
+
}
|
|
738
|
+
}
|
|
739
|
+
return {
|
|
740
|
+
type: "link",
|
|
741
|
+
title: null,
|
|
742
|
+
raw: cap[0],
|
|
743
|
+
text,
|
|
744
|
+
href,
|
|
745
|
+
tokens: [
|
|
746
|
+
{
|
|
747
|
+
type: "text",
|
|
748
|
+
raw: text,
|
|
749
|
+
text,
|
|
750
|
+
},
|
|
751
|
+
],
|
|
752
|
+
};
|
|
753
|
+
}
|
|
754
|
+
return undefined;
|
|
755
|
+
}
|
|
756
|
+
|
|
757
|
+
inlineText(src: string): Tokens["Text"] | undefined {
|
|
758
|
+
const cap = inline.text.exec(src);
|
|
759
|
+
if (!cap) return undefined;
|
|
760
|
+
|
|
761
|
+
let text;
|
|
762
|
+
if (this.lexer.state.inRawBlock) {
|
|
763
|
+
text = cap[0];
|
|
764
|
+
} else {
|
|
765
|
+
text = escape(cap[0]);
|
|
766
|
+
}
|
|
767
|
+
return {
|
|
768
|
+
type: "text",
|
|
769
|
+
raw: cap[0],
|
|
770
|
+
text,
|
|
771
|
+
};
|
|
772
|
+
}
|
|
773
|
+
}
|