pantsdown 2.3.2 → 2.3.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/package.json +11 -30
- package/src/css/styles.css +23 -0
- package/src/lexer.ts +74 -9
- package/src/renderer.ts +10 -7
- package/src/rules/block.ts +8 -7
- package/src/rules/inline.ts +56 -10
- package/src/rules/other.ts +22 -6
- package/src/tokenizer.ts +103 -55
- package/src/types.ts +5 -0
- package/src/utils.ts +105 -11
package/README.md
CHANGED
|
@@ -200,4 +200,4 @@ console.log(html, javascript);
|
|
|
200
200
|
Pantsdown is based on [Marked](https://github.com/markedjs/marked). Without their hard work,
|
|
201
201
|
Pantsdown would not exist.
|
|
202
202
|
|
|
203
|
-
Last synced with Marked [v18.0.
|
|
203
|
+
Last synced with Marked [v18.0.14](https://github.com/markedjs/marked/releases/tag/v18.0.14).
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pantsdown",
|
|
3
|
-
"version": "2.3.
|
|
3
|
+
"version": "2.3.4",
|
|
4
4
|
"description": "Markdown to 'GitHub HTML' parser",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"author": "wallpants",
|
|
@@ -27,7 +27,6 @@
|
|
|
27
27
|
}
|
|
28
28
|
},
|
|
29
29
|
"scripts": {
|
|
30
|
-
"commit": "cz",
|
|
31
30
|
"format": "oxfmt .",
|
|
32
31
|
"typecheck": "tsc",
|
|
33
32
|
"lint": "oxlint .",
|
|
@@ -37,40 +36,22 @@
|
|
|
37
36
|
},
|
|
38
37
|
"dependencies": {
|
|
39
38
|
"github-slugger": "^2.0.0",
|
|
40
|
-
"highlight.js": "^11.
|
|
41
|
-
"katex": "^0.18.
|
|
39
|
+
"highlight.js": "^11.12.0",
|
|
40
|
+
"katex": "^0.18.9"
|
|
42
41
|
},
|
|
43
42
|
"devDependencies": {
|
|
44
|
-
"@commitlint/cli": "^21.2.
|
|
45
|
-
"@commitlint/config-conventional": "^21.2.
|
|
46
|
-
"@
|
|
47
|
-
"@
|
|
48
|
-
"@types/bun": "^1.3.14",
|
|
43
|
+
"@commitlint/cli": "^21.2.3",
|
|
44
|
+
"@commitlint/config-conventional": "^21.2.3",
|
|
45
|
+
"@happy-dom/global-registrator": "^20.14.5",
|
|
46
|
+
"@types/bun": "^1.4.2",
|
|
49
47
|
"@types/katex": "^0.16.8",
|
|
50
|
-
"commitizen": "^4.3.2",
|
|
51
48
|
"husky": "^9.1.7",
|
|
52
|
-
"
|
|
53
|
-
"
|
|
54
|
-
"oxlint": "^
|
|
55
|
-
"
|
|
56
|
-
"semantic-release": "^25.0.8",
|
|
49
|
+
"oxfmt": "^0.71.0",
|
|
50
|
+
"oxlint": "^1.86.0",
|
|
51
|
+
"oxlint-tsgolint": "^7.0.2003",
|
|
52
|
+
"semantic-release": "^25.0.9",
|
|
57
53
|
"typescript": "7.0.2"
|
|
58
54
|
},
|
|
59
|
-
"commitlint": {
|
|
60
|
-
"extends": [
|
|
61
|
-
"@commitlint/config-conventional"
|
|
62
|
-
],
|
|
63
|
-
"rules": {
|
|
64
|
-
"body-max-line-length": [
|
|
65
|
-
0
|
|
66
|
-
]
|
|
67
|
-
}
|
|
68
|
-
},
|
|
69
|
-
"config": {
|
|
70
|
-
"commitizen": {
|
|
71
|
-
"path": "@commitlint/cz-commitlint"
|
|
72
|
-
}
|
|
73
|
-
},
|
|
74
55
|
"release": {
|
|
75
56
|
"branches": [
|
|
76
57
|
"main"
|
package/src/css/styles.css
CHANGED
|
@@ -336,6 +336,8 @@ html.pantsdown-mermaid-mod.pantsdown .mermaid-viewport {
|
|
|
336
336
|
.pantsdown a {
|
|
337
337
|
background-color: transparent;
|
|
338
338
|
color: var(--color-accent-fg);
|
|
339
|
+
/* explicit ua default: resets set `text-decoration: inherit` */
|
|
340
|
+
text-decoration: underline;
|
|
339
341
|
}
|
|
340
342
|
|
|
341
343
|
.pantsdown abbr[title] {
|
|
@@ -391,6 +393,9 @@ html.pantsdown-mermaid-mod.pantsdown .mermaid-viewport {
|
|
|
391
393
|
max-width: 100%;
|
|
392
394
|
box-sizing: content-box;
|
|
393
395
|
background-color: var(--color-canvas-default);
|
|
396
|
+
/* explicit ua defaults: resets set `display: block; vertical-align: middle` */
|
|
397
|
+
display: inline;
|
|
398
|
+
vertical-align: baseline;
|
|
394
399
|
}
|
|
395
400
|
|
|
396
401
|
.pantsdown code,
|
|
@@ -611,6 +616,24 @@ html.pantsdown-mermaid-mod.pantsdown .mermaid-viewport {
|
|
|
611
616
|
padding-left: 2em;
|
|
612
617
|
}
|
|
613
618
|
|
|
619
|
+
/* consumers may apply a css reset (e.g. tailwind preflight sets
|
|
620
|
+
`list-style: none`), so don't rely on user-agent defaults */
|
|
621
|
+
.pantsdown ul {
|
|
622
|
+
list-style-type: disc;
|
|
623
|
+
}
|
|
624
|
+
|
|
625
|
+
.pantsdown ul ul {
|
|
626
|
+
list-style-type: circle;
|
|
627
|
+
}
|
|
628
|
+
|
|
629
|
+
.pantsdown ul ul ul {
|
|
630
|
+
list-style-type: square;
|
|
631
|
+
}
|
|
632
|
+
|
|
633
|
+
.pantsdown ol {
|
|
634
|
+
list-style-type: decimal;
|
|
635
|
+
}
|
|
636
|
+
|
|
614
637
|
.pantsdown ol ol,
|
|
615
638
|
.pantsdown ul ol {
|
|
616
639
|
list-style-type: lower-roman;
|
package/src/lexer.ts
CHANGED
|
@@ -2,6 +2,7 @@ import { inline } from "./rules/inline.ts";
|
|
|
2
2
|
import { other } from "./rules/other.ts";
|
|
3
3
|
import { Tokenizer } from "./tokenizer.ts";
|
|
4
4
|
import { type Links, type SourceMap, type Token, type Tokens } from "./types.ts";
|
|
5
|
+
import { normalizeLabel } from "./utils.ts";
|
|
5
6
|
|
|
6
7
|
export class Lexer {
|
|
7
8
|
private tokenizer: Tokenizer;
|
|
@@ -14,6 +15,8 @@ export class Lexer {
|
|
|
14
15
|
state = {
|
|
15
16
|
inLink: false,
|
|
16
17
|
inRawBlock: false,
|
|
18
|
+
/** a link was produced in the inline run currently being scanned */
|
|
19
|
+
linkEmitted: false,
|
|
17
20
|
top: true,
|
|
18
21
|
};
|
|
19
22
|
|
|
@@ -35,6 +38,7 @@ export class Lexer {
|
|
|
35
38
|
this.state = {
|
|
36
39
|
inLink: false,
|
|
37
40
|
inRawBlock: false,
|
|
41
|
+
linkEmitted: false,
|
|
38
42
|
top: true,
|
|
39
43
|
};
|
|
40
44
|
|
|
@@ -215,6 +219,9 @@ export class Lexer {
|
|
|
215
219
|
if (lastParagraphClipped && lastToken?.type === "paragraph") {
|
|
216
220
|
lastToken.raw += (lastToken.raw.endsWith("\n") ? "" : "\n") + token.raw;
|
|
217
221
|
lastToken.text += "\n" + token.text;
|
|
222
|
+
if (lastToken.sourceMap && token.sourceMap) {
|
|
223
|
+
lastToken.sourceMap[1] = token.sourceMap[1];
|
|
224
|
+
}
|
|
218
225
|
this.inlineQueue.pop();
|
|
219
226
|
const lastInline = this.inlineQueue[this.inlineQueue.length - 1];
|
|
220
227
|
if (lastInline) lastInline.src = lastToken.text;
|
|
@@ -256,6 +263,42 @@ export class Lexer {
|
|
|
256
263
|
return tokens;
|
|
257
264
|
}
|
|
258
265
|
|
|
266
|
+
/**
|
|
267
|
+
* Does this link text hold a link already? An image does not count: an image
|
|
268
|
+
* may hold a link, a link may not.
|
|
269
|
+
*/
|
|
270
|
+
private linkInText(text: string): boolean {
|
|
271
|
+
if (!text.includes("[")) {
|
|
272
|
+
return false;
|
|
273
|
+
}
|
|
274
|
+
|
|
275
|
+
for (const match of text.matchAll(inline.blockSkip)) {
|
|
276
|
+
// blockSkip also matches code spans and html, and the `!` of an image is
|
|
277
|
+
// left out of the match, so read the character before it.
|
|
278
|
+
if (inline.link.test(match[0]) && text.charAt(match.index - 1) !== "!") {
|
|
279
|
+
return true;
|
|
280
|
+
}
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
for (const match of text.matchAll(inline.reflinkSearch)) {
|
|
284
|
+
const match0 = match[0];
|
|
285
|
+
const refStart = match0.lastIndexOf("[");
|
|
286
|
+
if (
|
|
287
|
+
match0.startsWith("!") ||
|
|
288
|
+
!Object.hasOwn(this.links, normalizeLabel(match0.slice(refStart + 1, -1)))
|
|
289
|
+
) {
|
|
290
|
+
continue;
|
|
291
|
+
}
|
|
292
|
+
// a candidate holding a link is not a link either, so it does not count
|
|
293
|
+
if (refStart > 1 && this.linkInText(match0.slice(1, refStart - 1))) {
|
|
294
|
+
continue;
|
|
295
|
+
}
|
|
296
|
+
return true;
|
|
297
|
+
}
|
|
298
|
+
|
|
299
|
+
return false;
|
|
300
|
+
}
|
|
301
|
+
|
|
259
302
|
/**
|
|
260
303
|
* Lexing/Compiling
|
|
261
304
|
*/
|
|
@@ -267,16 +310,38 @@ export class Lexer {
|
|
|
267
310
|
let keepPrevChar, prevChar;
|
|
268
311
|
|
|
269
312
|
// Mask out reflinks
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
313
|
+
if (src.includes("[")) {
|
|
314
|
+
const maskReflink = (match0: string): string => {
|
|
315
|
+
const refStart = match0.lastIndexOf("[");
|
|
316
|
+
if (!Object.hasOwn(this.links, normalizeLabel(match0.slice(refStart + 1, -1)))) {
|
|
317
|
+
return match0;
|
|
318
|
+
}
|
|
319
|
+
// CommonMark: "Links may not contain other links, at any level of
|
|
320
|
+
// nesting." A candidate whose text already holds one never becomes a
|
|
321
|
+
// link, so flattening the whole span would hide the emphasis that
|
|
322
|
+
// does still apply inside it. Mask the links it holds instead.
|
|
323
|
+
// Images are exempt: their text is flattened into an alt attribute.
|
|
324
|
+
if (refStart > 1 && !match0.startsWith("!")) {
|
|
325
|
+
const text = match0.slice(1, refStart - 1);
|
|
326
|
+
if (this.linkInText(text)) {
|
|
327
|
+
return (
|
|
328
|
+
"[" +
|
|
329
|
+
text.replace(inline.reflinkSearch, maskReflink) +
|
|
330
|
+
"][" +
|
|
331
|
+
"a".repeat(match0.length - refStart - 2) +
|
|
332
|
+
"]"
|
|
333
|
+
);
|
|
334
|
+
}
|
|
335
|
+
}
|
|
336
|
+
return "[" + "a".repeat(match0.length - 2) + "]";
|
|
337
|
+
};
|
|
338
|
+
maskedSrc = maskedSrc.replace(inline.reflinkSearch, maskReflink);
|
|
277
339
|
}
|
|
278
|
-
// Mask out escaped characters
|
|
279
|
-
|
|
340
|
+
// Mask out escaped characters.
|
|
341
|
+
// Every mask must keep the length it replaces: emStrong and del line
|
|
342
|
+
// maskedSrc up with src by slicing from the end. `anyPunctuation` matches
|
|
343
|
+
// unicode punctuation, so an escaped astral character is 3 code units.
|
|
344
|
+
maskedSrc = maskedSrc.replace(inline.anyPunctuation, (match0) => "+".repeat(match0.length));
|
|
280
345
|
|
|
281
346
|
// Mask out other blocks
|
|
282
347
|
maskedSrc = maskedSrc.replace(
|
package/src/renderer.ts
CHANGED
|
@@ -60,7 +60,8 @@ export class Renderer {
|
|
|
60
60
|
}
|
|
61
61
|
|
|
62
62
|
const language = lang && hljs.getLanguage(lang) ? lang : "plaintext";
|
|
63
|
-
|
|
63
|
+
// An empty code block has no content, so it must not gain a newline.
|
|
64
|
+
code = hljs.highlight(code ? code + "\n" : "", { language }).value;
|
|
64
65
|
code = `<code class="hljs language-${escape(language)}">${code}</code>`;
|
|
65
66
|
|
|
66
67
|
const result = `<pre style="position: relative;">` + code + `</pre>`;
|
|
@@ -234,15 +235,17 @@ export class Renderer {
|
|
|
234
235
|
return `<del>${this.parser.parseInline(tokens)}</del>`;
|
|
235
236
|
}
|
|
236
237
|
|
|
237
|
-
link({ href, title, tokens }: Tokens["Link"]): string {
|
|
238
|
-
|
|
238
|
+
link({ href, title, text, tokens, autolink }: Tokens["Link"]): string {
|
|
239
|
+
// References are not resolved inside an autolink, so every `&` there is
|
|
240
|
+
// literal. Elsewhere only an `&` that cannot start one needs escaping.
|
|
241
|
+
const parsedText = autolink ? escape(text, true) : this.parser.parseInline(tokens);
|
|
239
242
|
const cleanHref = cleanUrl(href);
|
|
240
243
|
if (cleanHref === null) {
|
|
241
|
-
return
|
|
244
|
+
return parsedText;
|
|
242
245
|
}
|
|
243
|
-
const attrs: HTMLAttrs = [["href", cleanHref]];
|
|
246
|
+
const attrs: HTMLAttrs = [["href", escape(cleanHref, autolink)]];
|
|
244
247
|
if (title) attrs.push(["title", escape(title)]);
|
|
245
|
-
return injectHtmlAttributes(`<a>${
|
|
248
|
+
return injectHtmlAttributes(`<a>${parsedText}</a>`, attrs);
|
|
246
249
|
}
|
|
247
250
|
|
|
248
251
|
image({ href, title, tokens }: Tokens["Image"]): string {
|
|
@@ -252,7 +255,7 @@ export class Renderer {
|
|
|
252
255
|
return escape(text);
|
|
253
256
|
}
|
|
254
257
|
const attrs: HTMLAttrs = [
|
|
255
|
-
["src", fixLocalImageHref(cleanHref, this.pantsdown.config)],
|
|
258
|
+
["src", escape(fixLocalImageHref(cleanHref, this.pantsdown.config))],
|
|
256
259
|
["alt", escape(text)],
|
|
257
260
|
];
|
|
258
261
|
if (title) attrs.push(["title", escape(title)]);
|
package/src/rules/block.ts
CHANGED
|
@@ -21,7 +21,7 @@ type BlockRuleNames =
|
|
|
21
21
|
|
|
22
22
|
export const label = /(?!\s*\])(?:\\[\s\S]|[^\[\]\\])+/;
|
|
23
23
|
|
|
24
|
-
const tag =
|
|
24
|
+
export const tag =
|
|
25
25
|
"address|article|aside|base|basefont|blockquote|body|caption" +
|
|
26
26
|
"|center|col|colgroup|dd|details|dialog|dir|div|dl|dt|fieldset|figcaption" +
|
|
27
27
|
"|figure|footer|form|frame|frameset|h[1-6]|head|header|hr|html|iframe" +
|
|
@@ -60,8 +60,8 @@ const block_html = edit(
|
|
|
60
60
|
"|<![A-Z][\\s\\S]*?(?:>[^\\n]*\\n*|$)" + // (4)
|
|
61
61
|
"|<!\\[CDATA\\[[\\s\\S]*?(?:\\]\\]>[^\\n]*\\n*|$)" + // (5)
|
|
62
62
|
"|</?(tag)(?: +|\\n|/?>)[\\s\\S]*?(?:(?:\\n[ \\t]*)+\\n|$)" + // (6)
|
|
63
|
-
"|<(?!script|pre|style|textarea)([a-z][
|
|
64
|
-
"|</(?!script|pre|style|textarea)[a-z][
|
|
63
|
+
"|<(?!script|pre|style|textarea)([a-z][a-z0-9-]*)(?:attribute)*? */?>(?=[ \\t]*(?:\\n|$))[\\s\\S]*?(?:(?:\\n[ \\t]*)+\\n|$)" + // (7) open tag
|
|
64
|
+
"|</(?!script|pre|style|textarea)[a-z][a-z0-9-]*\\s*>(?=[ \\t]*(?:\\n|$))[\\s\\S]*?(?:(?:\\n[ \\t]*)+\\n|$)" + // (7) closing tag
|
|
65
65
|
")",
|
|
66
66
|
"i",
|
|
67
67
|
)
|
|
@@ -72,13 +72,14 @@ const block_html = edit(
|
|
|
72
72
|
|
|
73
73
|
// upstream's lheadingGfm variant (we are GFM-only; the commonmark variant drops |table)
|
|
74
74
|
const block_lheading = edit(
|
|
75
|
-
/^(?!bull |blockCode|fences|blockquote|heading|html|table)((?:.|\n(?!\s*?\n|bull |
|
|
75
|
+
/^(?!bull |blockCode|fences|blockquote|heading|html|table)((?:.|\n(?!\s*?\n|bull |fences|blockquote|heading|hr|html|table))+?)\n {0,3}(=+|-+) *(?:\n+|$)/,
|
|
76
76
|
)
|
|
77
77
|
.replace(/bull/g, block_bullet) // lists can interrupt
|
|
78
|
-
.replace(/blockCode/g, /(?: {4}| {0,3}\t)/) // indented code
|
|
78
|
+
.replace(/blockCode/g, /(?: {4}| {0,3}\t)/) // indented code can start a block but cannot interrupt a paragraph
|
|
79
79
|
.replace(/fences/g, / {0,3}(?:`{3,}|~{3,})/) // fenced code blocks can interrupt
|
|
80
80
|
.replace(/blockquote/g, / {0,3}>/) // blockquote can interrupt
|
|
81
81
|
.replace(/heading/g, / {0,3}#{1,6}(?:\s|$)/) // ATX heading can interrupt
|
|
82
|
+
.replace(/hr/g, / {0,3}(?:(?:-[\t ]*){3,}|(?:_[ \t]*){3,}|(?:\*[ \t]*){3,})(?:\n+|$)/) // thematic break can interrupt
|
|
82
83
|
.replace(/html/g, / {0,3}<[^\n>]+>\n/) // block html can interrupt
|
|
83
84
|
.replace(/table/g, / {0,3}\|?(?:[:\- ]*\|)+[\:\- ]*\n/) // table can interrupt
|
|
84
85
|
.getRegex();
|
|
@@ -94,7 +95,7 @@ const block_table = edit(
|
|
|
94
95
|
.replace("heading", " {0,3}#{1,6}(?:\\s|$)")
|
|
95
96
|
.replace("blockquote", " {0,3}>")
|
|
96
97
|
.replace("code", "(?: {4}| {0,3}\\t)[^\\n]")
|
|
97
|
-
.replace("fences", " {0,3}(?:`{3,}(?=[^`\\n]
|
|
98
|
+
.replace("fences", " {0,3}(?:`{3,}(?=[^`\\n]*(?:\\n|$))|~~~)[^\\n]*(?:\\n|$)")
|
|
98
99
|
.replace("list", " {0,3}(?:[*+-]|1[.)])[ \\t]") // any bullet ends the table rows
|
|
99
100
|
.replace("html", "</?(?:tag)(?: +|\\n|/?>)|<(?:script|pre|style|textarea|!--)")
|
|
100
101
|
.replace("tag", tag) // tables can be interrupted by type (6) html blocks
|
|
@@ -107,7 +108,7 @@ const createParagraph = (listInterrupt: string) =>
|
|
|
107
108
|
.replace("|lheading", "") // setext headings don't interrupt commonmark paragraphs
|
|
108
109
|
.replace("table", block_table) // interrupt paragraphs with table
|
|
109
110
|
.replace("blockquote", " {0,3}>")
|
|
110
|
-
.replace("fences", " {0,3}(?:`{3,}(?=[^`\\n]
|
|
111
|
+
.replace("fences", " {0,3}(?:`{3,}(?=[^`\\n]*(?:\\n|$))|~~~)[^\\n]*(?:\\n|$)")
|
|
111
112
|
.replace("list", listInterrupt)
|
|
112
113
|
.replace("html", "</?(?:tag)(?: +|\\n|/?>)|<(?:script|pre|style|textarea|!--)")
|
|
113
114
|
.replace("tag", tag) // pars can be interrupted by type (6) html blocks
|
package/src/rules/inline.ts
CHANGED
|
@@ -39,12 +39,25 @@ const href = /<(?:\\.|[^\n<>\\])+>|[^ \t\n\x00-\x1f]+|(?=\))/;
|
|
|
39
39
|
const scheme = /[a-zA-Z][a-zA-Z0-9+.-]{1,31}/;
|
|
40
40
|
const comment = edit(blockComment).replace("(?:-->|$)", "-->").getRegex();
|
|
41
41
|
const attribute = /\s+[a-zA-Z:_][\w.:-]*(?:\s*=\s*"[^"]*"|\s*=\s*'[^']*'|\s*=\s*[^\s"'=<>`]+)?/;
|
|
42
|
+
// One matched pair of brackets, holding no bracket of its own.
|
|
43
|
+
const labelBrackets = /\[(?:\\[\s\S]|[^\[\]\\])*\]/;
|
|
44
|
+
// CommonMark lets the brackets in a link label nest to any depth, which a regex
|
|
45
|
+
// cannot follow, so it stops at a fixed one. Two levels is what the spec suite
|
|
46
|
+
// asks for: a third and a fourth flip no further example.
|
|
42
47
|
// codespan branches carry the #3918 ReDoS fix (`+(?!`) head, ``+(?=\]) tail)
|
|
43
|
-
const label =
|
|
44
|
-
/(?:\[(
|
|
48
|
+
const label = edit(
|
|
49
|
+
/(?:\[(?:brackets|\\[\s\S]|[^\[\]\\])*\]|\\[\s\S]|`+(?!`)[^`]*?`+(?!`)|``+(?=\])|[^\[\]\\`])*?/,
|
|
50
|
+
)
|
|
51
|
+
.replace("brackets", labelBrackets)
|
|
52
|
+
.getRegex();
|
|
45
53
|
const email =
|
|
46
54
|
/[a-zA-Z0-9.!#$%&'*+/=?^_`{|}~-]+(@)[a-zA-Z0-9](?:[a-zA-Z0-9-]{0,61}[a-zA-Z0-9])?(?:\.[a-zA-Z0-9](?:[a-zA-Z0-9-]{0,61}[a-zA-Z0-9])?)+(?![-_])/;
|
|
47
|
-
const extended_email = /[A-Za-z0-9._+-]+(@)[a-zA-Z0-9-_]+(?:\.[a-zA-Z0-9-_]*[a-zA-Z0-9])+(?![-
|
|
55
|
+
const extended_email = /[A-Za-z0-9._+-]+(@)[a-zA-Z0-9-_]+(?:\.[a-zA-Z0-9-_]*[a-zA-Z0-9])+(?![\w-])/;
|
|
56
|
+
// GFM protocol autolinks (`mailto:`/`xmpp:`); the email here has no `(@)` group,
|
|
57
|
+
// so the url tokenizer treats a match as a plain url (href = text)
|
|
58
|
+
const extended_email_protocol = edit(/(?:mailto:email|xmpp:email(?:\/[A-Za-z0-9@.]+)?)/)
|
|
59
|
+
.replace(/email/g, /[A-Za-z0-9._+-]+@[a-zA-Z0-9-_]+(?:\.[a-zA-Z0-9-_]*[a-zA-Z0-9])+(?![\w-])/)
|
|
60
|
+
.getRegex();
|
|
48
61
|
|
|
49
62
|
const inline_punctuation = edit(/^((?![*_])punctSpace)/, "u")
|
|
50
63
|
.replace(/punctSpace/g, _punctuationOrSpace)
|
|
@@ -106,8 +119,8 @@ const inline_autolink = edit(/^<(scheme:[^\s\x00-\x1f<>]*|email)>/)
|
|
|
106
119
|
|
|
107
120
|
const inline_tag = edit(
|
|
108
121
|
"^comment" +
|
|
109
|
-
"|^</[a-zA-Z][
|
|
110
|
-
"|^<[a-zA-Z][
|
|
122
|
+
"|^</[a-zA-Z][a-zA-Z0-9-]*\\s*>" + // self-closing tag
|
|
123
|
+
"|^<[a-zA-Z][a-zA-Z0-9-]*(?:attribute)*?\\s*/?>" + // open tag
|
|
111
124
|
"|^<\\?[\\s\\S]*?\\?>" + // processing instruction, e.g. <?php ?>
|
|
112
125
|
"|^<![a-zA-Z]+\\s[\\s\\S]*?>" + // declaration, e.g. <!DOCTYPE html>
|
|
113
126
|
"|^<!\\[CDATA\\[[\\s\\S]*?\\]\\]>", // CDATA section
|
|
@@ -133,9 +146,40 @@ const inline_nolink = edit(/^!?\[(ref)\](?:\[\])?/)
|
|
|
133
146
|
.replace("ref", blockLabel)
|
|
134
147
|
.getRegex();
|
|
135
148
|
|
|
149
|
+
// `reflink` and `nolink` are anchored, so the tokenizer tries each of them at a
|
|
150
|
+
// single position. `reflinkSearch` drops the anchors and runs with the global
|
|
151
|
+
// flag, which makes every '[' in the source a start position. A label crosses a
|
|
152
|
+
// bracket only by escaping it, and nothing caps how often it does, so a
|
|
153
|
+
// candidate that can never match still scans to the end of the source and the
|
|
154
|
+
// whole pass costs O(n^2).
|
|
155
|
+
//
|
|
156
|
+
// The labels below bound that scan. `boundedInlineLabel` limits how many
|
|
157
|
+
// escapes, code spans and nested brackets a link text may hold, leaving runs of
|
|
158
|
+
// ordinary characters unbounded so link text of any length is still found.
|
|
159
|
+
// `boundedBlockLabel` limits the label itself to 999 items, which is every
|
|
160
|
+
// label CommonMark allows: "A link label can have at most 999 characters
|
|
161
|
+
// inside the square brackets."
|
|
162
|
+
const boundedBlockLabel = /(?!\s*\])(?:\\[\s\S]|[^\[\]\\]){1,999}/;
|
|
163
|
+
const boundedInlineLabel = edit(
|
|
164
|
+
/(?:[^\[\]\\`]*(?:\[(?:brackets|\\[\s\S]|[^\[\]\\])*\]|\\[\s\S]|`+(?!`)[^`]*?`+(?!`)|``+(?=\]))){0,999}?[^\[\]\\`]*?/,
|
|
165
|
+
)
|
|
166
|
+
.replace("brackets", labelBrackets)
|
|
167
|
+
.getRegex();
|
|
168
|
+
|
|
136
169
|
const inline_reflinkSearch = edit("reflink|nolink(?!\\()", "g")
|
|
137
|
-
.replace(
|
|
138
|
-
|
|
170
|
+
.replace(
|
|
171
|
+
"reflink",
|
|
172
|
+
edit(/^!?\[(label)\]\[(ref)\]/)
|
|
173
|
+
.replace("label", boundedInlineLabel)
|
|
174
|
+
.replace("ref", boundedBlockLabel)
|
|
175
|
+
.getRegex(),
|
|
176
|
+
)
|
|
177
|
+
.replace(
|
|
178
|
+
"nolink",
|
|
179
|
+
edit(/^!?\[(ref)\](?:\[\])?/)
|
|
180
|
+
.replace("ref", boundedBlockLabel)
|
|
181
|
+
.getRegex(),
|
|
182
|
+
)
|
|
139
183
|
.getRegex();
|
|
140
184
|
|
|
141
185
|
const inline_escape = /^\\([!"#$%&'()*+,\-./:;<=>?@\[\]\\^_`{|}~])/;
|
|
@@ -165,9 +209,10 @@ const inline_delRDelim = edit(delRDelimCore, "gu")
|
|
|
165
209
|
.getRegex();
|
|
166
210
|
|
|
167
211
|
const inline_text = edit(
|
|
168
|
-
/^(`+|~+|[^`~])(?:(?=[`~])|(?= {2,}\n)|(?=[a-zA-Z0-9.!#$%&'*+\/=?_`{\|}~-]+@)|[\s\S]*?(?:(?=[\\<!\[`*~_$]|\b_|protocol:\/\/|www\.|$)|[^ ](?= {2,}\n)|[^a-zA-Z0-9.!#$%&'*+\/=?_`{\|}~-](?=[a-zA-Z0-9.!#$%&'*+\/=?_`{\|}~-]+@)))/,
|
|
212
|
+
/^(?:[^a-zA-Z0-9](?=emailProtocol)|(`+|~+|[^`~])(?:(?=[`~])|(?= {2,}\n)|(?=[a-zA-Z0-9.!#$%&'*+\/=?_`{\|}~-]+@)|[\s\S]*?(?:(?=[\\<!\[`*~_$]|\b_|protocol:\/\/|www\.|$)|[^ ](?= {2,}\n)|[^a-zA-Z0-9](?=emailProtocol)|[^a-zA-Z0-9.!#$%&'*+\/=?_`{\|}~-](?=[a-zA-Z0-9.!#$%&'*+\/=?_`{\|}~-]+@))))/,
|
|
169
213
|
)
|
|
170
214
|
.replace("protocol", /[hH][tT][tT][pP][sS]?|[fF][tT][pP]/)
|
|
215
|
+
.replace(/emailProtocol/g, /(?:mailto|xmpp):/)
|
|
171
216
|
.getRegex();
|
|
172
217
|
|
|
173
218
|
// DEVIATION from upstream `(?:[a-zA-Z0-9\-]+\.?)+`: the nested quantifier
|
|
@@ -176,8 +221,9 @@ const inline_text = edit(
|
|
|
176
221
|
// domain form matches the exact same language (fuzz-verified over 200k
|
|
177
222
|
// samples) — preserve it when porting upstream changes to this rule.
|
|
178
223
|
const inline_url = edit(
|
|
179
|
-
/^((?:protocol):\/\/|www\.)(?:[a-zA-Z0-9\-]+\.)*[a-zA-Z0-9\-]+\.?[^\s<]*|^email/,
|
|
224
|
+
/^emailProtocol|^((?:protocol):\/\/|www\.)(?:[a-zA-Z0-9\-]+\.)*[a-zA-Z0-9\-]+\.?[^\s<]*|^email/,
|
|
180
225
|
)
|
|
226
|
+
.replace("emailProtocol", extended_email_protocol)
|
|
181
227
|
.replace("protocol", /[fF][tT][pP]|[hH][tT][tT][pP][sS]?/)
|
|
182
228
|
.replace("email", extended_email)
|
|
183
229
|
.getRegex();
|
|
@@ -196,7 +242,7 @@ export const inline: Omit<Record<InlineRuleNames, RegExp>, "emStrong"> & {
|
|
|
196
242
|
anyPunctuation: inline_anyPunctuation,
|
|
197
243
|
emStrong: inline_emStrong,
|
|
198
244
|
code: /^(`+)([^`]|[^`][\s\S]*?[^`])\1(?!`)/,
|
|
199
|
-
br: /^( {2,}|\\)\n(?!\s*$)
|
|
245
|
+
br: /^( {2,}|\\)\n(?!\s*$)[ \t]*/,
|
|
200
246
|
delLDelim: inline_delLDelim,
|
|
201
247
|
delRDelim: inline_delRDelim,
|
|
202
248
|
text: inline_text,
|
package/src/rules/other.ts
CHANGED
|
@@ -1,9 +1,11 @@
|
|
|
1
|
+
import { tag } from "./block.ts";
|
|
2
|
+
|
|
1
3
|
/**
|
|
2
4
|
* Regexes that don't belong to the block or inline grammars.
|
|
3
|
-
* Names and values mirror marked's `other` rules object (src/rules.ts,
|
|
4
|
-
*
|
|
5
|
-
* future syncs stay diffable.
|
|
5
|
+
* Names and values mirror marked's `other` rules object (src/rules.ts, as of
|
|
6
|
+
* the "Last synced" version in the root README) so future syncs stay diffable.
|
|
6
7
|
*/
|
|
8
|
+
|
|
7
9
|
function cachedIndentRegex(createRegex: (indent: number) => RegExp) {
|
|
8
10
|
const cache: RegExp[] = [];
|
|
9
11
|
return (indent: number) => {
|
|
@@ -18,11 +20,13 @@ function cachedIndentRegex(createRegex: (indent: number) => RegExp) {
|
|
|
18
20
|
}
|
|
19
21
|
|
|
20
22
|
export const other = {
|
|
21
|
-
codeRemoveIndent: /^(?: {
|
|
23
|
+
codeRemoveIndent: /^(?: {0,3}\t| {1,4})/gm,
|
|
22
24
|
tabCharGlobal: /\t/g,
|
|
25
|
+
leadingSpaceTab: /^[ \t]+/,
|
|
23
26
|
outputLinkReplace: /\\([\[\]])/g,
|
|
24
27
|
indentCodeCompensation: /^(\s+)(?:```)/,
|
|
25
28
|
beginningSpace: /^\s+/,
|
|
29
|
+
endingSpaceTabChar: /[ \t]$/,
|
|
26
30
|
nonSpaceChar: /[^ ]/,
|
|
27
31
|
newLineCharGlobal: /\n/g,
|
|
28
32
|
multipleSpaceGlobal: /\s+/g,
|
|
@@ -42,11 +46,17 @@ export const other = {
|
|
|
42
46
|
tableAlignRight: /^ *-+: *$/,
|
|
43
47
|
tableAlignCenter: /^ *:-+: *$/,
|
|
44
48
|
tableAlignLeft: /^ *:-+ *$/,
|
|
49
|
+
// Pantsdown: setext heading continuation lines (see Tokenizer.lheading)
|
|
50
|
+
continuationIndent: /\n[ \t]+/g,
|
|
51
|
+
// Pantsdown: html block sourcemaps (see Tokenizer.html)
|
|
52
|
+
htmlOpenTagName: /^ {0,3}<([a-zA-Z][a-zA-Z0-9-]*)/,
|
|
53
|
+
htmlEndingCloseTagName: /<\/([a-zA-Z][a-zA-Z0-9-]*)\s*>$/,
|
|
45
54
|
startATag: /^<a /i,
|
|
46
55
|
endATag: /^<\/a>/i,
|
|
47
56
|
startPreScriptTag: /^<(pre|code|kbd|script)(\s|>)/i,
|
|
48
57
|
endPreScriptTag: /^<\/(pre|code|kbd|script)(\s|>)/i,
|
|
49
58
|
unicodeAlphaNumeric: /[\p{L}\p{N}]/u,
|
|
59
|
+
numericCharacterReference: /&#(?:(\d{1,7})|[Xx]([A-Fa-f0-9]{1,6}));/g,
|
|
50
60
|
escapeTest: /[&<>"']/,
|
|
51
61
|
escapeReplace: /[&<>"']/g,
|
|
52
62
|
escapeTestNoEncode: /[<>"']|&(?!(#\d{1,7}|#[Xx][a-fA-F0-9]{1,6}|\w+);)/,
|
|
@@ -66,14 +76,20 @@ export const other = {
|
|
|
66
76
|
),
|
|
67
77
|
hrRegex: cachedIndentRegex(
|
|
68
78
|
(indent: number) =>
|
|
69
|
-
new RegExp(`^ {0,${indent}}((?:- *){3,}|(?:_ *){3,}|(?:\\* *){3,})(?:\\n+|$)`),
|
|
79
|
+
new RegExp(`^ {0,${indent}}((?:-[ \t]*){3,}|(?:_[ \t]*){3,}|(?:\\*[ \t]*){3,})(?:\\n+|$)`),
|
|
70
80
|
),
|
|
71
81
|
fencesBeginRegex: cachedIndentRegex(
|
|
72
82
|
(indent: number) => new RegExp(`^ {0,${indent}}(?:\`\`\`|~~~)`),
|
|
73
83
|
),
|
|
74
84
|
headingBeginRegex: cachedIndentRegex((indent: number) => new RegExp(`^ {0,${indent}}#`)),
|
|
85
|
+
// a list item ends where a paragraph would be interrupted, so this mirrors the
|
|
86
|
+
// html start conditions in the paragraph rule; type 7 is excluded there
|
|
75
87
|
htmlBeginRegex: cachedIndentRegex(
|
|
76
|
-
(indent: number) =>
|
|
88
|
+
(indent: number) =>
|
|
89
|
+
new RegExp(
|
|
90
|
+
`^ {0,${indent}}(?:</?(?:${tag})(?: +|$|/?>)|<(?:script|pre|style|textarea|!--))`,
|
|
91
|
+
"i",
|
|
92
|
+
),
|
|
77
93
|
),
|
|
78
94
|
blockquoteBeginRegex: cachedIndentRegex((indent: number) => new RegExp(`^ {0,${indent}}>`)),
|
|
79
95
|
};
|
package/src/tokenizer.ts
CHANGED
|
@@ -5,9 +5,12 @@ import { other } from "./rules/other.ts";
|
|
|
5
5
|
import { type Links, type Token, type Tokens } from "./types.ts";
|
|
6
6
|
import {
|
|
7
7
|
ALERTS,
|
|
8
|
+
decodeNumericCharacterReferences,
|
|
8
9
|
expandTabs,
|
|
9
10
|
findClosingBracket,
|
|
10
11
|
indentCodeCompensation,
|
|
12
|
+
isLabelEndInsideToken,
|
|
13
|
+
normalizeLabel,
|
|
11
14
|
outputLink,
|
|
12
15
|
rtrim,
|
|
13
16
|
splitCells,
|
|
@@ -19,7 +22,7 @@ import {
|
|
|
19
22
|
*/
|
|
20
23
|
export class Tokenizer {
|
|
21
24
|
private lexer: Lexer;
|
|
22
|
-
pendingHtmlClose:
|
|
25
|
+
pendingHtmlClose: { tag: string; token: Tokens["HTML"] }[] = [];
|
|
23
26
|
|
|
24
27
|
constructor(lexer: Lexer) {
|
|
25
28
|
this.lexer = lexer;
|
|
@@ -77,8 +80,8 @@ export class Tokenizer {
|
|
|
77
80
|
// remove trailing #s
|
|
78
81
|
if (text.endsWith("#")) {
|
|
79
82
|
const trimmed = rtrim(text, "#");
|
|
80
|
-
if (!trimmed ||
|
|
81
|
-
// CommonMark requires space before trailing #s
|
|
83
|
+
if (!trimmed || other.endingSpaceTabChar.test(trimmed)) {
|
|
84
|
+
// CommonMark requires a space or tab before trailing #s
|
|
82
85
|
text = trimmed.trim();
|
|
83
86
|
}
|
|
84
87
|
}
|
|
@@ -167,13 +170,27 @@ export class Tokenizer {
|
|
|
167
170
|
} else if (lastToken?.type === "blockquote") {
|
|
168
171
|
// include continuation in nested blockquote
|
|
169
172
|
const oldToken = lastToken;
|
|
170
|
-
|
|
173
|
+
// The continuation lines belong to the same nesting frame as the
|
|
174
|
+
// nested blockquote, which already had one '>' marker stripped, so
|
|
175
|
+
// strip one marker from them too. Otherwise a restated marker after a
|
|
176
|
+
// lazy line is re-parsed as a spurious deeper blockquote.
|
|
177
|
+
const continuation = lines.join("\n");
|
|
178
|
+
const newText =
|
|
179
|
+
oldToken.raw + "\n" + continuation.replace(other.blockquoteSetextReplace2, "");
|
|
171
180
|
// re-lexing the nested blockquote restarts at its first source line
|
|
172
181
|
this.lexer.line = startLine + raw.split("\n").length - oldToken.raw.split("\n").length;
|
|
173
182
|
const newToken = this.blockquote(newText)!;
|
|
174
183
|
tokens[tokens.length - 1] = newToken;
|
|
175
184
|
|
|
176
|
-
|
|
185
|
+
// Only include continuation lines the nested blockquote actually
|
|
186
|
+
// consumed. Unconsumed trailing lines must stay out of `raw` so the
|
|
187
|
+
// lexer can still tokenize them (paragraph, next blockquote, etc.).
|
|
188
|
+
const leftover = newText.substring(newToken.raw.length).replace(/^\n/, "");
|
|
189
|
+
const leftoverCount = leftover ? leftover.split("\n").length : 0;
|
|
190
|
+
const consumed = leftoverCount ? lines.slice(0, -leftoverCount) : lines;
|
|
191
|
+
if (consumed.length > 0) {
|
|
192
|
+
raw = `${raw}\n${consumed.join("\n")}`;
|
|
193
|
+
}
|
|
177
194
|
text = text.substring(0, text.length - oldToken.text.length) + newToken.text;
|
|
178
195
|
break;
|
|
179
196
|
} else if (lastToken?.type === "list") {
|
|
@@ -270,19 +287,23 @@ export class Tokenizer {
|
|
|
270
287
|
raw = cap[0];
|
|
271
288
|
src = src.substring(raw.length);
|
|
272
289
|
|
|
273
|
-
|
|
290
|
+
const firstLine = cap[2]!.split("\n", 1)[0]!;
|
|
291
|
+
const bulletIndent = cap[1]!.length;
|
|
292
|
+
let line = firstLine.replace(other.leadingSpaceTab, (whitespace) =>
|
|
293
|
+
expandTabs(whitespace, bulletIndent),
|
|
294
|
+
);
|
|
274
295
|
let nextLine = src.split("\n", 1)[0] ?? "";
|
|
275
296
|
|
|
276
297
|
let blankLine = !line.trim();
|
|
277
298
|
|
|
278
299
|
let indent = 0;
|
|
279
300
|
if (blankLine) {
|
|
280
|
-
indent =
|
|
301
|
+
indent = bulletIndent + 1;
|
|
281
302
|
} else {
|
|
282
303
|
indent = line.search(other.nonSpaceChar); // Find first non-space char
|
|
283
304
|
indent = indent > 4 ? 1 : indent; // Treat indented code blocks (> 4 spaces) as having only 1 indent
|
|
284
305
|
itemContents = line.slice(indent);
|
|
285
|
-
indent +=
|
|
306
|
+
indent += bulletIndent;
|
|
286
307
|
}
|
|
287
308
|
|
|
288
309
|
if (blankLine && other.blankLine.test(nextLine)) {
|
|
@@ -304,7 +325,9 @@ export class Tokenizer {
|
|
|
304
325
|
while (src) {
|
|
305
326
|
const rawLine = src.split("\n", 1)[0] ?? "";
|
|
306
327
|
nextLine = rawLine;
|
|
307
|
-
const nextLineWithoutTabs = nextLine.replace(other.
|
|
328
|
+
const nextLineWithoutTabs = nextLine.replace(other.leadingSpaceTab, (whitespace) =>
|
|
329
|
+
whitespace.replace(other.tabCharGlobal, " "),
|
|
330
|
+
);
|
|
308
331
|
|
|
309
332
|
// End list item if found code fences
|
|
310
333
|
if (fencesBeginRegex.test(nextLine)) {
|
|
@@ -416,25 +439,37 @@ export class Tokenizer {
|
|
|
416
439
|
// save/restore top: blockTokens resets it to true on exit, and a nested list
|
|
417
440
|
// lexed with a stale top=true would advance the sourcemap line counter
|
|
418
441
|
const top = this.lexer.state.top;
|
|
442
|
+
// First pass: tokenize items and finalize list.loose from spacers before placing checkboxes
|
|
419
443
|
for (const item of list.items) {
|
|
420
444
|
this.lexer.state.top = false;
|
|
421
445
|
item.tokens = this.lexer.blockTokens(item.text, []);
|
|
422
446
|
|
|
447
|
+
if (!list.loose) {
|
|
448
|
+
// Check if list should be loose
|
|
449
|
+
const spacers = item.tokens.filter((t) => t.type === "space");
|
|
450
|
+
const hasMultipleLineBreaks =
|
|
451
|
+
spacers.length > 0 && spacers.some((t) => other.anyLine.test(t.raw));
|
|
452
|
+
|
|
453
|
+
list.loose = hasMultipleLineBreaks;
|
|
454
|
+
}
|
|
455
|
+
}
|
|
456
|
+
this.lexer.state.top = top;
|
|
457
|
+
|
|
458
|
+
// Second pass: place task checkboxes using the final list.loose
|
|
459
|
+
for (const item of list.items) {
|
|
423
460
|
const itemToken = item.tokens[0];
|
|
424
461
|
if (item.task && (itemToken?.type === "text" || itemToken?.type === "paragraph")) {
|
|
425
462
|
// Remove checkbox markdown from item tokens
|
|
426
463
|
item.text = item.text.replace(other.listReplaceTask, "");
|
|
427
464
|
itemToken.raw = itemToken.raw.replace(other.listReplaceTask, "");
|
|
428
465
|
itemToken.text = itemToken.text.replace(other.listReplaceTask, "");
|
|
429
|
-
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
|
|
436
|
-
}
|
|
437
|
-
}
|
|
466
|
+
// DEVIATION: upstream strips the last queued src that looks like a task,
|
|
467
|
+
// which can be a later paragraph of the same item (`- [ ] a\n\n [ ] b`).
|
|
468
|
+
// The item token's own queue entry shares its tokens array.
|
|
469
|
+
const queued = this.lexer.inlineQueue.findLast(
|
|
470
|
+
(entry) => entry.tokens === itemToken.tokens,
|
|
471
|
+
);
|
|
472
|
+
if (queued) queued.src = queued.src.replace(other.listReplaceTask, "");
|
|
438
473
|
|
|
439
474
|
const taskRaw = other.listTaskCheckbox.exec(item.raw);
|
|
440
475
|
if (taskRaw) {
|
|
@@ -470,17 +505,7 @@ export class Tokenizer {
|
|
|
470
505
|
} else if (item.task) {
|
|
471
506
|
item.task = false;
|
|
472
507
|
}
|
|
473
|
-
|
|
474
|
-
if (!list.loose) {
|
|
475
|
-
// Check if list should be loose
|
|
476
|
-
const spacers = item.tokens.filter((t) => t.type === "space");
|
|
477
|
-
const hasMultipleLineBreaks =
|
|
478
|
-
spacers.length > 0 && spacers.some((t) => other.anyLine.test(t.raw));
|
|
479
|
-
|
|
480
|
-
list.loose = hasMultipleLineBreaks;
|
|
481
|
-
}
|
|
482
508
|
}
|
|
483
|
-
this.lexer.state.top = top;
|
|
484
509
|
|
|
485
510
|
// Set all items to loose if list is loose
|
|
486
511
|
if (list.loose) {
|
|
@@ -573,26 +598,26 @@ export class Tokenizer {
|
|
|
573
598
|
* and update their sourceMap once they're closed.
|
|
574
599
|
*/
|
|
575
600
|
|
|
576
|
-
const
|
|
577
|
-
|
|
578
|
-
|
|
579
|
-
|
|
580
|
-
|
|
581
|
-
|
|
582
|
-
|
|
583
|
-
|
|
584
|
-
|
|
585
|
-
|
|
586
|
-
|
|
587
|
-
|
|
588
|
-
|
|
589
|
-
|
|
590
|
-
if (updateToken.sourceMap?.[1] && token.sourceMap?.[1]) {
|
|
591
|
-
updateToken.sourceMap[1] = token.sourceMap[1];
|
|
592
|
-
}
|
|
593
|
-
this.pendingHtmlClose.shift();
|
|
594
|
-
}
|
|
601
|
+
const openTag = other.htmlOpenTagName.exec(raw)?.[1]?.toLowerCase();
|
|
602
|
+
const closeTag = other.htmlEndingCloseTagName.exec(raw.trimEnd())?.[1]?.toLowerCase();
|
|
603
|
+
|
|
604
|
+
// a token ending in the closing tag of a pending html token closes it, even
|
|
605
|
+
// when it opens other tags itself (` <dt>a</dt>\n</dl>`); one that closes
|
|
606
|
+
// the tag it opens (`<div>a</div>`) is self-contained
|
|
607
|
+
const index =
|
|
608
|
+
closeTag === undefined || closeTag === openTag
|
|
609
|
+
? -1
|
|
610
|
+
: this.pendingHtmlClose.findIndex((pending) => pending.tag === closeTag);
|
|
611
|
+
const pending = this.pendingHtmlClose[index];
|
|
612
|
+
if (pending) {
|
|
613
|
+
if (pending.token.sourceMap && token.sourceMap) {
|
|
614
|
+
pending.token.sourceMap[1] = token.sourceMap[1];
|
|
595
615
|
}
|
|
616
|
+
// tags opened after the one just closed can no longer be closed
|
|
617
|
+
this.pendingHtmlClose.splice(0, index + 1);
|
|
618
|
+
} else if (openTag && openTag !== closeTag) {
|
|
619
|
+
// first in last out
|
|
620
|
+
this.pendingHtmlClose.unshift({ tag: openTag, token });
|
|
596
621
|
}
|
|
597
622
|
|
|
598
623
|
return token;
|
|
@@ -602,7 +627,7 @@ export class Tokenizer {
|
|
|
602
627
|
const cap = block.def.exec(src);
|
|
603
628
|
if (!cap) return undefined;
|
|
604
629
|
|
|
605
|
-
const tag = cap[1]
|
|
630
|
+
const tag = normalizeLabel(cap[1]!).replace(other.multipleSpaceGlobal, " ");
|
|
606
631
|
const href = cap[2]
|
|
607
632
|
? cap[2].replace(other.hrefBrackets, "$1").replace(inline.anyPunctuation, "$1")
|
|
608
633
|
: "";
|
|
@@ -687,7 +712,9 @@ export class Tokenizer {
|
|
|
687
712
|
const cap = block.lheading.exec(src);
|
|
688
713
|
if (!cap) return undefined;
|
|
689
714
|
|
|
690
|
-
|
|
715
|
+
// DEVIATION: upstream keeps continuation lines' indentation; CommonMark (and
|
|
716
|
+
// GitHub, whose heading slugs would otherwise gain a dash per space) strips it
|
|
717
|
+
const text = cap[1]!.trim().replace(other.continuationIndent, "\n");
|
|
691
718
|
const raw = rtrim(cap[0], "\n");
|
|
692
719
|
return {
|
|
693
720
|
type: "heading",
|
|
@@ -766,6 +793,10 @@ export class Tokenizer {
|
|
|
766
793
|
const cap = inline.link.exec(src);
|
|
767
794
|
if (!cap) return undefined;
|
|
768
795
|
|
|
796
|
+
if (isLabelEndInsideToken(src, cap[1]!, cap[0].startsWith("!") ? 2 : 1)) {
|
|
797
|
+
return;
|
|
798
|
+
}
|
|
799
|
+
|
|
769
800
|
const trimmedUrl = cap[2]!.trim();
|
|
770
801
|
if (trimmedUrl.startsWith("<")) {
|
|
771
802
|
// commonmark requires matching angle brackets
|
|
@@ -818,8 +849,12 @@ export class Tokenizer {
|
|
|
818
849
|
): Tokens["Link"] | Tokens["Image"] | Tokens["Text"] | undefined {
|
|
819
850
|
let cap;
|
|
820
851
|
if ((cap = inline.reflink.exec(src)) ?? (cap = inline.nolink.exec(src))) {
|
|
852
|
+
if (isLabelEndInsideToken(src, cap[1]!, cap[0].startsWith("!") ? 2 : 1)) {
|
|
853
|
+
return;
|
|
854
|
+
}
|
|
855
|
+
|
|
821
856
|
const linkStr = (cap[2] ?? cap[1])!.replace(/\s+/g, " ");
|
|
822
|
-
const link = links[linkStr
|
|
857
|
+
const link = links[normalizeLabel(linkStr)];
|
|
823
858
|
if (!link) {
|
|
824
859
|
const text = cap[0].charAt(0);
|
|
825
860
|
return {
|
|
@@ -855,9 +890,12 @@ export class Tokenizer {
|
|
|
855
890
|
delimTotal = lLength,
|
|
856
891
|
midDelimTotal = 0;
|
|
857
892
|
|
|
858
|
-
const
|
|
859
|
-
|
|
860
|
-
|
|
893
|
+
const delimChar = match[0][0];
|
|
894
|
+
// A mid-run opener (for example the second star of an unmatched `**`) must
|
|
895
|
+
// only pair with a delimiter that can only close, otherwise it steals the
|
|
896
|
+
// opener of a later span (`**a*b*c` must be `**a<em>b</em>c`).
|
|
897
|
+
const midRun = prevChar === delimChar;
|
|
898
|
+
const endReg = delimChar === "*" ? inline.emStrong.rDelimAst : inline.emStrong.rDelimUnd;
|
|
861
899
|
endReg.lastIndex = 0;
|
|
862
900
|
|
|
863
901
|
// Clip maskedSrc to same section of string as src (move to lexer?)
|
|
@@ -880,6 +918,11 @@ export class Tokenizer {
|
|
|
880
918
|
midDelimTotal += rLength;
|
|
881
919
|
continue; // CommonMark Emphasis Rules 9-10
|
|
882
920
|
}
|
|
921
|
+
if (midRun) {
|
|
922
|
+
// A mid-run opener cannot close against an ambiguous delimiter that
|
|
923
|
+
// can also open; that delimiter opens its own emphasis span instead.
|
|
924
|
+
break;
|
|
925
|
+
}
|
|
883
926
|
}
|
|
884
927
|
|
|
885
928
|
delimTotal -= rLength;
|
|
@@ -1032,6 +1075,7 @@ export class Tokenizer {
|
|
|
1032
1075
|
raw: cap[0],
|
|
1033
1076
|
text,
|
|
1034
1077
|
href,
|
|
1078
|
+
autolink: true,
|
|
1035
1079
|
tokens: [
|
|
1036
1080
|
{
|
|
1037
1081
|
type: "text",
|
|
@@ -1069,6 +1113,7 @@ export class Tokenizer {
|
|
|
1069
1113
|
raw: cap[0],
|
|
1070
1114
|
text,
|
|
1071
1115
|
href,
|
|
1116
|
+
autolink: true,
|
|
1072
1117
|
tokens: [
|
|
1073
1118
|
{
|
|
1074
1119
|
type: "text",
|
|
@@ -1085,11 +1130,14 @@ export class Tokenizer {
|
|
|
1085
1130
|
const cap = inline.text.exec(src);
|
|
1086
1131
|
if (!cap) return undefined;
|
|
1087
1132
|
|
|
1133
|
+
const escaped = this.lexer.state.inRawBlock;
|
|
1088
1134
|
return {
|
|
1089
1135
|
type: "text",
|
|
1090
1136
|
raw: cap[0],
|
|
1091
|
-
|
|
1092
|
-
|
|
1137
|
+
// raw HTML keeps whatever it was written with, everywhere else a numeric
|
|
1138
|
+
// character reference stands for the character itself
|
|
1139
|
+
text: escaped ? cap[0] : decodeNumericCharacterReferences(cap[0]),
|
|
1140
|
+
escaped,
|
|
1093
1141
|
};
|
|
1094
1142
|
}
|
|
1095
1143
|
|
package/src/types.ts
CHANGED
|
@@ -141,6 +141,11 @@ export interface Tokens extends Record<string, BaseToken> {
|
|
|
141
141
|
href: string;
|
|
142
142
|
title: string | null;
|
|
143
143
|
tokens: Token[];
|
|
144
|
+
/**
|
|
145
|
+
* Set for autolinks and extended (GFM) urls, where character references are
|
|
146
|
+
* not resolved, so the destination and text are literal.
|
|
147
|
+
*/
|
|
148
|
+
autolink?: boolean;
|
|
144
149
|
};
|
|
145
150
|
Image: {
|
|
146
151
|
type: "image";
|
package/src/utils.ts
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { type Lexer } from "./lexer.ts";
|
|
2
|
+
import { inline } from "./rules/inline.ts";
|
|
2
3
|
import { other } from "./rules/other.ts";
|
|
3
4
|
import { type HTMLAttrs, type PantsdownConfig, type SourceMap, type Tokens } from "./types.ts";
|
|
4
5
|
|
|
@@ -154,6 +155,24 @@ export function fixLocalImageHref(href: string, config: PantsdownConfig): string
|
|
|
154
155
|
}
|
|
155
156
|
}
|
|
156
157
|
|
|
158
|
+
/**
|
|
159
|
+
* Numeric character references are recognized outside code and are equivalent to the
|
|
160
|
+
* character they name. Values that are zero, out of range, or a surrogate become the
|
|
161
|
+
* replacement character.
|
|
162
|
+
*/
|
|
163
|
+
export function decodeNumericCharacterReferences(text: string) {
|
|
164
|
+
return text.replace(
|
|
165
|
+
other.numericCharacterReference,
|
|
166
|
+
(_, dec: string | undefined, hex: string | undefined) => {
|
|
167
|
+
const code = dec === undefined ? Number.parseInt(hex!, 16) : Number.parseInt(dec, 10);
|
|
168
|
+
if (code === 0 || code > 0x10ffff || (code >= 0xd800 && code <= 0xdfff)) {
|
|
169
|
+
return "\uFFFD";
|
|
170
|
+
}
|
|
171
|
+
return String.fromCodePoint(code);
|
|
172
|
+
},
|
|
173
|
+
);
|
|
174
|
+
}
|
|
175
|
+
|
|
157
176
|
export function cleanUrl(href: string) {
|
|
158
177
|
try {
|
|
159
178
|
href = encodeURI(href).replace(other.percentDecode, "%");
|
|
@@ -252,6 +271,20 @@ export function trimTrailingBlankLines(str: string) {
|
|
|
252
271
|
return lines.slice(0, end + 1).join("\n");
|
|
253
272
|
}
|
|
254
273
|
|
|
274
|
+
/**
|
|
275
|
+
* Normalizes a link label so definitions and references can be matched.
|
|
276
|
+
* CommonMark asks for a Unicode case fold, which `toLowerCase()` does not
|
|
277
|
+
* reach: `ẞ` lowercases to `ß` and so never meets `SS`. Round-tripping
|
|
278
|
+
* through upper case does, as in commonmark.js; the final `toLowerCase()`
|
|
279
|
+
* keeps the folded label lower case, the form `def.tag` has always used.
|
|
280
|
+
*/
|
|
281
|
+
export function normalizeLabel(label: string) {
|
|
282
|
+
// The spec also asks for leading and trailing spaces, tabs and line endings
|
|
283
|
+
// to be stripped. Doing it here keeps every call site in agreement: the
|
|
284
|
+
// definition and the reference have to normalize to the same key.
|
|
285
|
+
return label.trim().toLowerCase().toUpperCase().toLowerCase();
|
|
286
|
+
}
|
|
287
|
+
|
|
255
288
|
export function findClosingBracket(str: string, b: string) {
|
|
256
289
|
if (!b[1] || !str.includes(b[1])) {
|
|
257
290
|
return -1;
|
|
@@ -281,22 +314,43 @@ export function outputLink(
|
|
|
281
314
|
link: Pick<Tokens["Link"], "href" | "title">,
|
|
282
315
|
raw: string,
|
|
283
316
|
lexer: Lexer,
|
|
284
|
-
): Tokens["Link"] | Tokens["Image"] {
|
|
317
|
+
): Tokens["Link"] | Tokens["Image"] | undefined {
|
|
285
318
|
const href = link.href;
|
|
286
319
|
const title = link.title || null;
|
|
287
320
|
const text = cap[1]?.replace(other.outputLinkReplace, "$1") ?? "";
|
|
321
|
+
const isImage = cap[0]?.charAt(0) === "!";
|
|
288
322
|
|
|
289
323
|
lexer.state.inLink = true;
|
|
290
|
-
const
|
|
291
|
-
|
|
324
|
+
const outerLinkEmitted = lexer.state.linkEmitted;
|
|
325
|
+
const outerInRawBlock = lexer.state.inRawBlock;
|
|
326
|
+
lexer.state.linkEmitted = false;
|
|
327
|
+
const tokens = lexer.inlineTokens(text);
|
|
328
|
+
// widen: TS keeps the `= false` narrowing across the inlineTokens call that sets it
|
|
329
|
+
const textHasLink = lexer.state.linkEmitted as boolean;
|
|
330
|
+
lexer.state.linkEmitted = outerLinkEmitted;
|
|
331
|
+
lexer.state.inLink = false;
|
|
332
|
+
|
|
333
|
+
if (!isImage) {
|
|
334
|
+
// CommonMark: "Links may not contain other links, at any level of nesting."
|
|
335
|
+
// Bail so the caller falls through to text and the inner link is the one kept.
|
|
336
|
+
// Images are exempt: their text is flattened into an alt attribute.
|
|
337
|
+
if (textHasLink) {
|
|
338
|
+
// these tokens are discarded, so undo the raw-block state they opened;
|
|
339
|
+
// leaving it set would suppress escaping for the text that is re-scanned
|
|
340
|
+
lexer.state.inRawBlock = outerInRawBlock;
|
|
341
|
+
return;
|
|
342
|
+
}
|
|
343
|
+
lexer.state.linkEmitted = true;
|
|
344
|
+
}
|
|
345
|
+
|
|
346
|
+
return {
|
|
347
|
+
type: isImage ? "image" : "link",
|
|
292
348
|
raw,
|
|
293
349
|
href,
|
|
294
350
|
title,
|
|
295
351
|
text,
|
|
296
|
-
tokens
|
|
352
|
+
tokens,
|
|
297
353
|
};
|
|
298
|
-
lexer.state.inLink = false;
|
|
299
|
-
return token;
|
|
300
354
|
}
|
|
301
355
|
|
|
302
356
|
export function indentCodeCompensation(raw: string, text: string) {
|
|
@@ -318,15 +372,55 @@ export function indentCodeCompensation(raw: string, text: string) {
|
|
|
318
372
|
|
|
319
373
|
const [indentInNode] = matchIndentInNode;
|
|
320
374
|
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
return node;
|
|
375
|
+
// Up to the fence's own indentation is removed from each line, so a line
|
|
376
|
+
// indented less than the fence loses whatever indentation it has.
|
|
377
|
+
return node.slice(Math.min(indentInNode.length, indentToCode?.length ?? 0));
|
|
326
378
|
})
|
|
327
379
|
.join("\n");
|
|
328
380
|
}
|
|
329
381
|
|
|
382
|
+
/**
|
|
383
|
+
* Does the link label end inside a raw token (html tag or autolink) that starts
|
|
384
|
+
* within it? Such a token takes precedence over the link, so its closing `]`
|
|
385
|
+
* does not close the label.
|
|
386
|
+
*/
|
|
387
|
+
export function isLabelEndInsideToken(src: string, label: string, labelStart: number) {
|
|
388
|
+
if (!label.includes("<")) {
|
|
389
|
+
return false;
|
|
390
|
+
}
|
|
391
|
+
|
|
392
|
+
for (let i = 0; i < label.length; i++) {
|
|
393
|
+
if (label[i] === "\\") {
|
|
394
|
+
i++;
|
|
395
|
+
continue;
|
|
396
|
+
}
|
|
397
|
+
|
|
398
|
+
if (label[i] === "`") {
|
|
399
|
+
const code = inline.code.exec(label.slice(i));
|
|
400
|
+
if (code) {
|
|
401
|
+
i += code[0].length - 1;
|
|
402
|
+
continue;
|
|
403
|
+
}
|
|
404
|
+
}
|
|
405
|
+
|
|
406
|
+
if (label[i] !== "<") {
|
|
407
|
+
continue;
|
|
408
|
+
}
|
|
409
|
+
|
|
410
|
+
const tokenSrc = src.slice(labelStart + i);
|
|
411
|
+
const token = inline.tag.exec(tokenSrc) ?? inline.autolink.exec(tokenSrc);
|
|
412
|
+
if (!token) {
|
|
413
|
+
continue;
|
|
414
|
+
}
|
|
415
|
+
|
|
416
|
+
if (token[0].length > label.length - i) {
|
|
417
|
+
return true;
|
|
418
|
+
}
|
|
419
|
+
i += token[0].length - 1;
|
|
420
|
+
}
|
|
421
|
+
return false;
|
|
422
|
+
}
|
|
423
|
+
|
|
330
424
|
function makeAlertRegex(type: string) {
|
|
331
425
|
return new RegExp(`^(?:\\[\\!${type.toUpperCase()}\\]|[\\*]{2}${type}[\\*]{2})[ \\t]*\\n?`);
|
|
332
426
|
}
|