pantsdown 2.3.3 → 2.3.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -200,4 +200,4 @@ console.log(html, javascript);
200
200
  Pantsdown is based on [Marked](https://github.com/markedjs/marked). Without their hard work,
201
201
  Pantsdown would not exist.
202
202
 
203
- Last synced with Marked [v18.0.7](https://github.com/markedjs/marked/releases/tag/v18.0.7).
203
+ Last synced with Marked [v18.0.14](https://github.com/markedjs/marked/releases/tag/v18.0.14).
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pantsdown",
3
- "version": "2.3.3",
3
+ "version": "2.3.4",
4
4
  "description": "Markdown to 'GitHub HTML' parser",
5
5
  "license": "MIT",
6
6
  "author": "wallpants",
@@ -27,7 +27,6 @@
27
27
  }
28
28
  },
29
29
  "scripts": {
30
- "commit": "cz",
31
30
  "format": "oxfmt .",
32
31
  "typecheck": "tsc",
33
32
  "lint": "oxlint .",
@@ -37,40 +36,22 @@
37
36
  },
38
37
  "dependencies": {
39
38
  "github-slugger": "^2.0.0",
40
- "highlight.js": "^11.11.1",
41
- "katex": "^0.18.1"
39
+ "highlight.js": "^11.12.0",
40
+ "katex": "^0.18.9"
42
41
  },
43
42
  "devDependencies": {
44
- "@commitlint/cli": "^21.2.1",
45
- "@commitlint/config-conventional": "^21.2.0",
46
- "@commitlint/cz-commitlint": "^21.2.0",
47
- "@happy-dom/global-registrator": "^20.11.1",
48
- "@types/bun": "^1.3.14",
43
+ "@commitlint/cli": "^21.2.3",
44
+ "@commitlint/config-conventional": "^21.2.3",
45
+ "@happy-dom/global-registrator": "^20.14.5",
46
+ "@types/bun": "^1.4.2",
49
47
  "@types/katex": "^0.16.8",
50
- "commitizen": "^4.3.2",
51
48
  "husky": "^9.1.7",
52
- "inquirer": "^12.11.1",
53
- "oxfmt": "^0.62.0",
54
- "oxlint": "^1.77.0",
55
- "oxlint-tsgolint": "^7.0.2001",
56
- "semantic-release": "^25.0.8",
49
+ "oxfmt": "^0.71.0",
50
+ "oxlint": "^1.86.0",
51
+ "oxlint-tsgolint": "^7.0.2003",
52
+ "semantic-release": "^25.0.9",
57
53
  "typescript": "7.0.2"
58
54
  },
59
- "commitlint": {
60
- "extends": [
61
- "@commitlint/config-conventional"
62
- ],
63
- "rules": {
64
- "body-max-line-length": [
65
- 0
66
- ]
67
- }
68
- },
69
- "config": {
70
- "commitizen": {
71
- "path": "@commitlint/cz-commitlint"
72
- }
73
- },
74
55
  "release": {
75
56
  "branches": [
76
57
  "main"
package/src/lexer.ts CHANGED
@@ -2,6 +2,7 @@ import { inline } from "./rules/inline.ts";
2
2
  import { other } from "./rules/other.ts";
3
3
  import { Tokenizer } from "./tokenizer.ts";
4
4
  import { type Links, type SourceMap, type Token, type Tokens } from "./types.ts";
5
+ import { normalizeLabel } from "./utils.ts";
5
6
 
6
7
  export class Lexer {
7
8
  private tokenizer: Tokenizer;
@@ -14,6 +15,8 @@ export class Lexer {
14
15
  state = {
15
16
  inLink: false,
16
17
  inRawBlock: false,
18
+ /** a link was produced in the inline run currently being scanned */
19
+ linkEmitted: false,
17
20
  top: true,
18
21
  };
19
22
 
@@ -35,6 +38,7 @@ export class Lexer {
35
38
  this.state = {
36
39
  inLink: false,
37
40
  inRawBlock: false,
41
+ linkEmitted: false,
38
42
  top: true,
39
43
  };
40
44
 
@@ -215,6 +219,9 @@ export class Lexer {
215
219
  if (lastParagraphClipped && lastToken?.type === "paragraph") {
216
220
  lastToken.raw += (lastToken.raw.endsWith("\n") ? "" : "\n") + token.raw;
217
221
  lastToken.text += "\n" + token.text;
222
+ if (lastToken.sourceMap && token.sourceMap) {
223
+ lastToken.sourceMap[1] = token.sourceMap[1];
224
+ }
218
225
  this.inlineQueue.pop();
219
226
  const lastInline = this.inlineQueue[this.inlineQueue.length - 1];
220
227
  if (lastInline) lastInline.src = lastToken.text;
@@ -256,6 +263,42 @@ export class Lexer {
256
263
  return tokens;
257
264
  }
258
265
 
266
+ /**
267
+ * Does this link text hold a link already? An image does not count: an image
268
+ * may hold a link, a link may not.
269
+ */
270
+ private linkInText(text: string): boolean {
271
+ if (!text.includes("[")) {
272
+ return false;
273
+ }
274
+
275
+ for (const match of text.matchAll(inline.blockSkip)) {
276
+ // blockSkip also matches code spans and html, and the `!` of an image is
277
+ // left out of the match, so read the character before it.
278
+ if (inline.link.test(match[0]) && text.charAt(match.index - 1) !== "!") {
279
+ return true;
280
+ }
281
+ }
282
+
283
+ for (const match of text.matchAll(inline.reflinkSearch)) {
284
+ const match0 = match[0];
285
+ const refStart = match0.lastIndexOf("[");
286
+ if (
287
+ match0.startsWith("!") ||
288
+ !Object.hasOwn(this.links, normalizeLabel(match0.slice(refStart + 1, -1)))
289
+ ) {
290
+ continue;
291
+ }
292
+ // a candidate holding a link is not a link either, so it does not count
293
+ if (refStart > 1 && this.linkInText(match0.slice(1, refStart - 1))) {
294
+ continue;
295
+ }
296
+ return true;
297
+ }
298
+
299
+ return false;
300
+ }
301
+
259
302
  /**
260
303
  * Lexing/Compiling
261
304
  */
@@ -267,16 +310,38 @@ export class Lexer {
267
310
  let keepPrevChar, prevChar;
268
311
 
269
312
  // Mask out reflinks
270
- const links = Object.keys(this.links);
271
- if (links.length > 0) {
272
- maskedSrc = maskedSrc.replace(inline.reflinkSearch, (match0) =>
273
- links.includes(match0.slice(match0.lastIndexOf("[") + 1, -1))
274
- ? "[" + "a".repeat(match0.length - 2) + "]"
275
- : match0,
276
- );
313
+ if (src.includes("[")) {
314
+ const maskReflink = (match0: string): string => {
315
+ const refStart = match0.lastIndexOf("[");
316
+ if (!Object.hasOwn(this.links, normalizeLabel(match0.slice(refStart + 1, -1)))) {
317
+ return match0;
318
+ }
319
+ // CommonMark: "Links may not contain other links, at any level of
320
+ // nesting." A candidate whose text already holds one never becomes a
321
+ // link, so flattening the whole span would hide the emphasis that
322
+ // does still apply inside it. Mask the links it holds instead.
323
+ // Images are exempt: their text is flattened into an alt attribute.
324
+ if (refStart > 1 && !match0.startsWith("!")) {
325
+ const text = match0.slice(1, refStart - 1);
326
+ if (this.linkInText(text)) {
327
+ return (
328
+ "[" +
329
+ text.replace(inline.reflinkSearch, maskReflink) +
330
+ "][" +
331
+ "a".repeat(match0.length - refStart - 2) +
332
+ "]"
333
+ );
334
+ }
335
+ }
336
+ return "[" + "a".repeat(match0.length - 2) + "]";
337
+ };
338
+ maskedSrc = maskedSrc.replace(inline.reflinkSearch, maskReflink);
277
339
  }
278
- // Mask out escaped characters
279
- maskedSrc = maskedSrc.replace(inline.anyPunctuation, "++");
340
+ // Mask out escaped characters.
341
+ // Every mask must keep the length it replaces: emStrong and del line
342
+ // maskedSrc up with src by slicing from the end. `anyPunctuation` matches
343
+ // unicode punctuation, so an escaped astral character is 3 code units.
344
+ maskedSrc = maskedSrc.replace(inline.anyPunctuation, (match0) => "+".repeat(match0.length));
280
345
 
281
346
  // Mask out other blocks
282
347
  maskedSrc = maskedSrc.replace(
package/src/renderer.ts CHANGED
@@ -60,7 +60,8 @@ export class Renderer {
60
60
  }
61
61
 
62
62
  const language = lang && hljs.getLanguage(lang) ? lang : "plaintext";
63
- code = hljs.highlight(code + "\n", { language }).value;
63
+ // An empty code block has no content, so it must not gain a newline.
64
+ code = hljs.highlight(code ? code + "\n" : "", { language }).value;
64
65
  code = `<code class="hljs language-${escape(language)}">${code}</code>`;
65
66
 
66
67
  const result = `<pre style="position: relative;">` + code + `</pre>`;
@@ -234,15 +235,17 @@ export class Renderer {
234
235
  return `<del>${this.parser.parseInline(tokens)}</del>`;
235
236
  }
236
237
 
237
- link({ href, title, tokens }: Tokens["Link"]): string {
238
- const text = this.parser.parseInline(tokens);
238
+ link({ href, title, text, tokens, autolink }: Tokens["Link"]): string {
239
+ // References are not resolved inside an autolink, so every `&` there is
240
+ // literal. Elsewhere only an `&` that cannot start one needs escaping.
241
+ const parsedText = autolink ? escape(text, true) : this.parser.parseInline(tokens);
239
242
  const cleanHref = cleanUrl(href);
240
243
  if (cleanHref === null) {
241
- return text;
244
+ return parsedText;
242
245
  }
243
- const attrs: HTMLAttrs = [["href", cleanHref]];
246
+ const attrs: HTMLAttrs = [["href", escape(cleanHref, autolink)]];
244
247
  if (title) attrs.push(["title", escape(title)]);
245
- return injectHtmlAttributes(`<a>${text}</a>`, attrs);
248
+ return injectHtmlAttributes(`<a>${parsedText}</a>`, attrs);
246
249
  }
247
250
 
248
251
  image({ href, title, tokens }: Tokens["Image"]): string {
@@ -252,7 +255,7 @@ export class Renderer {
252
255
  return escape(text);
253
256
  }
254
257
  const attrs: HTMLAttrs = [
255
- ["src", fixLocalImageHref(cleanHref, this.pantsdown.config)],
258
+ ["src", escape(fixLocalImageHref(cleanHref, this.pantsdown.config))],
256
259
  ["alt", escape(text)],
257
260
  ];
258
261
  if (title) attrs.push(["title", escape(title)]);
@@ -21,7 +21,7 @@ type BlockRuleNames =
21
21
 
22
22
  export const label = /(?!\s*\])(?:\\[\s\S]|[^\[\]\\])+/;
23
23
 
24
- const tag =
24
+ export const tag =
25
25
  "address|article|aside|base|basefont|blockquote|body|caption" +
26
26
  "|center|col|colgroup|dd|details|dialog|dir|div|dl|dt|fieldset|figcaption" +
27
27
  "|figure|footer|form|frame|frameset|h[1-6]|head|header|hr|html|iframe" +
@@ -60,8 +60,8 @@ const block_html = edit(
60
60
  "|<![A-Z][\\s\\S]*?(?:>[^\\n]*\\n*|$)" + // (4)
61
61
  "|<!\\[CDATA\\[[\\s\\S]*?(?:\\]\\]>[^\\n]*\\n*|$)" + // (5)
62
62
  "|</?(tag)(?: +|\\n|/?>)[\\s\\S]*?(?:(?:\\n[ \\t]*)+\\n|$)" + // (6)
63
- "|<(?!script|pre|style|textarea)([a-z][\\w-]*)(?:attribute)*? */?>(?=[ \\t]*(?:\\n|$))[\\s\\S]*?(?:(?:\\n[ \\t]*)+\\n|$)" + // (7) open tag
64
- "|</(?!script|pre|style|textarea)[a-z][\\w-]*\\s*>(?=[ \\t]*(?:\\n|$))[\\s\\S]*?(?:(?:\\n[ \\t]*)+\\n|$)" + // (7) closing tag
63
+ "|<(?!script|pre|style|textarea)([a-z][a-z0-9-]*)(?:attribute)*? */?>(?=[ \\t]*(?:\\n|$))[\\s\\S]*?(?:(?:\\n[ \\t]*)+\\n|$)" + // (7) open tag
64
+ "|</(?!script|pre|style|textarea)[a-z][a-z0-9-]*\\s*>(?=[ \\t]*(?:\\n|$))[\\s\\S]*?(?:(?:\\n[ \\t]*)+\\n|$)" + // (7) closing tag
65
65
  ")",
66
66
  "i",
67
67
  )
@@ -72,13 +72,14 @@ const block_html = edit(
72
72
 
73
73
  // upstream's lheadingGfm variant (we are GFM-only; the commonmark variant drops |table)
74
74
  const block_lheading = edit(
75
- /^(?!bull |blockCode|fences|blockquote|heading|html|table)((?:.|\n(?!\s*?\n|bull |blockCode|fences|blockquote|heading|html|table))+?)\n {0,3}(=+|-+) *(?:\n+|$)/,
75
+ /^(?!bull |blockCode|fences|blockquote|heading|html|table)((?:.|\n(?!\s*?\n|bull |fences|blockquote|heading|hr|html|table))+?)\n {0,3}(=+|-+) *(?:\n+|$)/,
76
76
  )
77
77
  .replace(/bull/g, block_bullet) // lists can interrupt
78
- .replace(/blockCode/g, /(?: {4}| {0,3}\t)/) // indented code blocks can interrupt
78
+ .replace(/blockCode/g, /(?: {4}| {0,3}\t)/) // indented code can start a block but cannot interrupt a paragraph
79
79
  .replace(/fences/g, / {0,3}(?:`{3,}|~{3,})/) // fenced code blocks can interrupt
80
80
  .replace(/blockquote/g, / {0,3}>/) // blockquote can interrupt
81
81
  .replace(/heading/g, / {0,3}#{1,6}(?:\s|$)/) // ATX heading can interrupt
82
+ .replace(/hr/g, / {0,3}(?:(?:-[\t ]*){3,}|(?:_[ \t]*){3,}|(?:\*[ \t]*){3,})(?:\n+|$)/) // thematic break can interrupt
82
83
  .replace(/html/g, / {0,3}<[^\n>]+>\n/) // block html can interrupt
83
84
  .replace(/table/g, / {0,3}\|?(?:[:\- ]*\|)+[\:\- ]*\n/) // table can interrupt
84
85
  .getRegex();
@@ -94,7 +95,7 @@ const block_table = edit(
94
95
  .replace("heading", " {0,3}#{1,6}(?:\\s|$)")
95
96
  .replace("blockquote", " {0,3}>")
96
97
  .replace("code", "(?: {4}| {0,3}\\t)[^\\n]")
97
- .replace("fences", " {0,3}(?:`{3,}(?=[^`\\n]*\\n)|~~~)[^\\n]*\\n")
98
+ .replace("fences", " {0,3}(?:`{3,}(?=[^`\\n]*(?:\\n|$))|~~~)[^\\n]*(?:\\n|$)")
98
99
  .replace("list", " {0,3}(?:[*+-]|1[.)])[ \\t]") // any bullet ends the table rows
99
100
  .replace("html", "</?(?:tag)(?: +|\\n|/?>)|<(?:script|pre|style|textarea|!--)")
100
101
  .replace("tag", tag) // tables can be interrupted by type (6) html blocks
@@ -107,7 +108,7 @@ const createParagraph = (listInterrupt: string) =>
107
108
  .replace("|lheading", "") // setext headings don't interrupt commonmark paragraphs
108
109
  .replace("table", block_table) // interrupt paragraphs with table
109
110
  .replace("blockquote", " {0,3}>")
110
- .replace("fences", " {0,3}(?:`{3,}(?=[^`\\n]*\\n)|~~~)[^\\n]*\\n")
111
+ .replace("fences", " {0,3}(?:`{3,}(?=[^`\\n]*(?:\\n|$))|~~~)[^\\n]*(?:\\n|$)")
111
112
  .replace("list", listInterrupt)
112
113
  .replace("html", "</?(?:tag)(?: +|\\n|/?>)|<(?:script|pre|style|textarea|!--)")
113
114
  .replace("tag", tag) // pars can be interrupted by type (6) html blocks
@@ -39,12 +39,25 @@ const href = /<(?:\\.|[^\n<>\\])+>|[^ \t\n\x00-\x1f]+|(?=\))/;
39
39
  const scheme = /[a-zA-Z][a-zA-Z0-9+.-]{1,31}/;
40
40
  const comment = edit(blockComment).replace("(?:-->|$)", "-->").getRegex();
41
41
  const attribute = /\s+[a-zA-Z:_][\w.:-]*(?:\s*=\s*"[^"]*"|\s*=\s*'[^']*'|\s*=\s*[^\s"'=<>`]+)?/;
42
+ // One matched pair of brackets, holding no bracket of its own.
43
+ const labelBrackets = /\[(?:\\[\s\S]|[^\[\]\\])*\]/;
44
+ // CommonMark lets the brackets in a link label nest to any depth, which a regex
45
+ // cannot follow, so it stops at a fixed one. Two levels is what the spec suite
46
+ // asks for: a third and a fourth flip no further example.
42
47
  // codespan branches carry the #3918 ReDoS fix (`+(?!`) head, ``+(?=\]) tail)
43
- const label =
44
- /(?:\[(?:\\[\s\S]|[^\[\]\\])*\]|\\[\s\S]|`+(?!`)[^`]*?`+(?!`)|``+(?=\])|[^\[\]\\`])*?/;
48
+ const label = edit(
49
+ /(?:\[(?:brackets|\\[\s\S]|[^\[\]\\])*\]|\\[\s\S]|`+(?!`)[^`]*?`+(?!`)|``+(?=\])|[^\[\]\\`])*?/,
50
+ )
51
+ .replace("brackets", labelBrackets)
52
+ .getRegex();
45
53
  const email =
46
54
  /[a-zA-Z0-9.!#$%&'*+/=?^_`{|}~-]+(@)[a-zA-Z0-9](?:[a-zA-Z0-9-]{0,61}[a-zA-Z0-9])?(?:\.[a-zA-Z0-9](?:[a-zA-Z0-9-]{0,61}[a-zA-Z0-9])?)+(?![-_])/;
47
- const extended_email = /[A-Za-z0-9._+-]+(@)[a-zA-Z0-9-_]+(?:\.[a-zA-Z0-9-_]*[a-zA-Z0-9])+(?![-_])/;
55
+ const extended_email = /[A-Za-z0-9._+-]+(@)[a-zA-Z0-9-_]+(?:\.[a-zA-Z0-9-_]*[a-zA-Z0-9])+(?![\w-])/;
56
+ // GFM protocol autolinks (`mailto:`/`xmpp:`); the email here has no `(@)` group,
57
+ // so the url tokenizer treats a match as a plain url (href = text)
58
+ const extended_email_protocol = edit(/(?:mailto:email|xmpp:email(?:\/[A-Za-z0-9@.]+)?)/)
59
+ .replace(/email/g, /[A-Za-z0-9._+-]+@[a-zA-Z0-9-_]+(?:\.[a-zA-Z0-9-_]*[a-zA-Z0-9])+(?![\w-])/)
60
+ .getRegex();
48
61
 
49
62
  const inline_punctuation = edit(/^((?![*_])punctSpace)/, "u")
50
63
  .replace(/punctSpace/g, _punctuationOrSpace)
@@ -106,8 +119,8 @@ const inline_autolink = edit(/^<(scheme:[^\s\x00-\x1f<>]*|email)>/)
106
119
 
107
120
  const inline_tag = edit(
108
121
  "^comment" +
109
- "|^</[a-zA-Z][\\w:-]*\\s*>" + // self-closing tag
110
- "|^<[a-zA-Z][\\w-]*(?:attribute)*?\\s*/?>" + // open tag
122
+ "|^</[a-zA-Z][a-zA-Z0-9-]*\\s*>" + // self-closing tag
123
+ "|^<[a-zA-Z][a-zA-Z0-9-]*(?:attribute)*?\\s*/?>" + // open tag
111
124
  "|^<\\?[\\s\\S]*?\\?>" + // processing instruction, e.g. <?php ?>
112
125
  "|^<![a-zA-Z]+\\s[\\s\\S]*?>" + // declaration, e.g. <!DOCTYPE html>
113
126
  "|^<!\\[CDATA\\[[\\s\\S]*?\\]\\]>", // CDATA section
@@ -133,9 +146,40 @@ const inline_nolink = edit(/^!?\[(ref)\](?:\[\])?/)
133
146
  .replace("ref", blockLabel)
134
147
  .getRegex();
135
148
 
149
+ // `reflink` and `nolink` are anchored, so the tokenizer tries each of them at a
150
+ // single position. `reflinkSearch` drops the anchors and runs with the global
151
+ // flag, which makes every '[' in the source a start position. A label crosses a
152
+ // bracket only by escaping it, and nothing caps how often it does, so a
153
+ // candidate that can never match still scans to the end of the source and the
154
+ // whole pass costs O(n^2).
155
+ //
156
+ // The labels below bound that scan. `boundedInlineLabel` limits how many
157
+ // escapes, code spans and nested brackets a link text may hold, leaving runs of
158
+ // ordinary characters unbounded so link text of any length is still found.
159
+ // `boundedBlockLabel` limits the label itself to 999 items, which is every
160
+ // label CommonMark allows: "A link label can have at most 999 characters
161
+ // inside the square brackets."
162
+ const boundedBlockLabel = /(?!\s*\])(?:\\[\s\S]|[^\[\]\\]){1,999}/;
163
+ const boundedInlineLabel = edit(
164
+ /(?:[^\[\]\\`]*(?:\[(?:brackets|\\[\s\S]|[^\[\]\\])*\]|\\[\s\S]|`+(?!`)[^`]*?`+(?!`)|``+(?=\]))){0,999}?[^\[\]\\`]*?/,
165
+ )
166
+ .replace("brackets", labelBrackets)
167
+ .getRegex();
168
+
136
169
  const inline_reflinkSearch = edit("reflink|nolink(?!\\()", "g")
137
- .replace("reflink", inline_reflink)
138
- .replace("nolink", inline_nolink)
170
+ .replace(
171
+ "reflink",
172
+ edit(/^!?\[(label)\]\[(ref)\]/)
173
+ .replace("label", boundedInlineLabel)
174
+ .replace("ref", boundedBlockLabel)
175
+ .getRegex(),
176
+ )
177
+ .replace(
178
+ "nolink",
179
+ edit(/^!?\[(ref)\](?:\[\])?/)
180
+ .replace("ref", boundedBlockLabel)
181
+ .getRegex(),
182
+ )
139
183
  .getRegex();
140
184
 
141
185
  const inline_escape = /^\\([!"#$%&'()*+,\-./:;<=>?@\[\]\\^_`{|}~])/;
@@ -165,9 +209,10 @@ const inline_delRDelim = edit(delRDelimCore, "gu")
165
209
  .getRegex();
166
210
 
167
211
  const inline_text = edit(
168
- /^(`+|~+|[^`~])(?:(?=[`~])|(?= {2,}\n)|(?=[a-zA-Z0-9.!#$%&'*+\/=?_`{\|}~-]+@)|[\s\S]*?(?:(?=[\\<!\[`*~_$]|\b_|protocol:\/\/|www\.|$)|[^ ](?= {2,}\n)|[^a-zA-Z0-9.!#$%&'*+\/=?_`{\|}~-](?=[a-zA-Z0-9.!#$%&'*+\/=?_`{\|}~-]+@)))/,
212
+ /^(?:[^a-zA-Z0-9](?=emailProtocol)|(`+|~+|[^`~])(?:(?=[`~])|(?= {2,}\n)|(?=[a-zA-Z0-9.!#$%&'*+\/=?_`{\|}~-]+@)|[\s\S]*?(?:(?=[\\<!\[`*~_$]|\b_|protocol:\/\/|www\.|$)|[^ ](?= {2,}\n)|[^a-zA-Z0-9](?=emailProtocol)|[^a-zA-Z0-9.!#$%&'*+\/=?_`{\|}~-](?=[a-zA-Z0-9.!#$%&'*+\/=?_`{\|}~-]+@))))/,
169
213
  )
170
214
  .replace("protocol", /[hH][tT][tT][pP][sS]?|[fF][tT][pP]/)
215
+ .replace(/emailProtocol/g, /(?:mailto|xmpp):/)
171
216
  .getRegex();
172
217
 
173
218
  // DEVIATION from upstream `(?:[a-zA-Z0-9\-]+\.?)+`: the nested quantifier
@@ -176,8 +221,9 @@ const inline_text = edit(
176
221
  // domain form matches the exact same language (fuzz-verified over 200k
177
222
  // samples) — preserve it when porting upstream changes to this rule.
178
223
  const inline_url = edit(
179
- /^((?:protocol):\/\/|www\.)(?:[a-zA-Z0-9\-]+\.)*[a-zA-Z0-9\-]+\.?[^\s<]*|^email/,
224
+ /^emailProtocol|^((?:protocol):\/\/|www\.)(?:[a-zA-Z0-9\-]+\.)*[a-zA-Z0-9\-]+\.?[^\s<]*|^email/,
180
225
  )
226
+ .replace("emailProtocol", extended_email_protocol)
181
227
  .replace("protocol", /[fF][tT][pP]|[hH][tT][tT][pP][sS]?/)
182
228
  .replace("email", extended_email)
183
229
  .getRegex();
@@ -196,7 +242,7 @@ export const inline: Omit<Record<InlineRuleNames, RegExp>, "emStrong"> & {
196
242
  anyPunctuation: inline_anyPunctuation,
197
243
  emStrong: inline_emStrong,
198
244
  code: /^(`+)([^`]|[^`][\s\S]*?[^`])\1(?!`)/,
199
- br: /^( {2,}|\\)\n(?!\s*$)/,
245
+ br: /^( {2,}|\\)\n(?!\s*$)[ \t]*/,
200
246
  delLDelim: inline_delLDelim,
201
247
  delRDelim: inline_delRDelim,
202
248
  text: inline_text,
@@ -1,9 +1,11 @@
1
+ import { tag } from "./block.ts";
2
+
1
3
  /**
2
4
  * Regexes that don't belong to the block or inline grammars.
3
- * Names and values mirror marked's `other` rules object (src/rules.ts,
4
- * currently marked v18.0.7 — see "Last synced" in the root README) so
5
- * future syncs stay diffable.
5
+ * Names and values mirror marked's `other` rules object (src/rules.ts, as of
6
+ * the "Last synced" version in the root README) so future syncs stay diffable.
6
7
  */
8
+
7
9
  function cachedIndentRegex(createRegex: (indent: number) => RegExp) {
8
10
  const cache: RegExp[] = [];
9
11
  return (indent: number) => {
@@ -18,11 +20,13 @@ function cachedIndentRegex(createRegex: (indent: number) => RegExp) {
18
20
  }
19
21
 
20
22
  export const other = {
21
- codeRemoveIndent: /^(?: {1,4}| {0,3}\t)/gm,
23
+ codeRemoveIndent: /^(?: {0,3}\t| {1,4})/gm,
22
24
  tabCharGlobal: /\t/g,
25
+ leadingSpaceTab: /^[ \t]+/,
23
26
  outputLinkReplace: /\\([\[\]])/g,
24
27
  indentCodeCompensation: /^(\s+)(?:```)/,
25
28
  beginningSpace: /^\s+/,
29
+ endingSpaceTabChar: /[ \t]$/,
26
30
  nonSpaceChar: /[^ ]/,
27
31
  newLineCharGlobal: /\n/g,
28
32
  multipleSpaceGlobal: /\s+/g,
@@ -42,11 +46,17 @@ export const other = {
42
46
  tableAlignRight: /^ *-+: *$/,
43
47
  tableAlignCenter: /^ *:-+: *$/,
44
48
  tableAlignLeft: /^ *:-+ *$/,
49
+ // Pantsdown: setext heading continuation lines (see Tokenizer.lheading)
50
+ continuationIndent: /\n[ \t]+/g,
51
+ // Pantsdown: html block sourcemaps (see Tokenizer.html)
52
+ htmlOpenTagName: /^ {0,3}<([a-zA-Z][a-zA-Z0-9-]*)/,
53
+ htmlEndingCloseTagName: /<\/([a-zA-Z][a-zA-Z0-9-]*)\s*>$/,
45
54
  startATag: /^<a /i,
46
55
  endATag: /^<\/a>/i,
47
56
  startPreScriptTag: /^<(pre|code|kbd|script)(\s|>)/i,
48
57
  endPreScriptTag: /^<\/(pre|code|kbd|script)(\s|>)/i,
49
58
  unicodeAlphaNumeric: /[\p{L}\p{N}]/u,
59
+ numericCharacterReference: /&#(?:(\d{1,7})|[Xx]([A-Fa-f0-9]{1,6}));/g,
50
60
  escapeTest: /[&<>"']/,
51
61
  escapeReplace: /[&<>"']/g,
52
62
  escapeTestNoEncode: /[<>"']|&(?!(#\d{1,7}|#[Xx][a-fA-F0-9]{1,6}|\w+);)/,
@@ -66,14 +76,20 @@ export const other = {
66
76
  ),
67
77
  hrRegex: cachedIndentRegex(
68
78
  (indent: number) =>
69
- new RegExp(`^ {0,${indent}}((?:- *){3,}|(?:_ *){3,}|(?:\\* *){3,})(?:\\n+|$)`),
79
+ new RegExp(`^ {0,${indent}}((?:-[ \t]*){3,}|(?:_[ \t]*){3,}|(?:\\*[ \t]*){3,})(?:\\n+|$)`),
70
80
  ),
71
81
  fencesBeginRegex: cachedIndentRegex(
72
82
  (indent: number) => new RegExp(`^ {0,${indent}}(?:\`\`\`|~~~)`),
73
83
  ),
74
84
  headingBeginRegex: cachedIndentRegex((indent: number) => new RegExp(`^ {0,${indent}}#`)),
85
+ // a list item ends where a paragraph would be interrupted, so this mirrors the
86
+ // html start conditions in the paragraph rule; type 7 is excluded there
75
87
  htmlBeginRegex: cachedIndentRegex(
76
- (indent: number) => new RegExp(`^ {0,${indent}}<(?:[a-z].*>|!--)`, "i"),
88
+ (indent: number) =>
89
+ new RegExp(
90
+ `^ {0,${indent}}(?:</?(?:${tag})(?: +|$|/?>)|<(?:script|pre|style|textarea|!--))`,
91
+ "i",
92
+ ),
77
93
  ),
78
94
  blockquoteBeginRegex: cachedIndentRegex((indent: number) => new RegExp(`^ {0,${indent}}>`)),
79
95
  };
package/src/tokenizer.ts CHANGED
@@ -5,9 +5,12 @@ import { other } from "./rules/other.ts";
5
5
  import { type Links, type Token, type Tokens } from "./types.ts";
6
6
  import {
7
7
  ALERTS,
8
+ decodeNumericCharacterReferences,
8
9
  expandTabs,
9
10
  findClosingBracket,
10
11
  indentCodeCompensation,
12
+ isLabelEndInsideToken,
13
+ normalizeLabel,
11
14
  outputLink,
12
15
  rtrim,
13
16
  splitCells,
@@ -19,7 +22,7 @@ import {
19
22
  */
20
23
  export class Tokenizer {
21
24
  private lexer: Lexer;
22
- pendingHtmlClose: [tag: string, index: number][] = [];
25
+ pendingHtmlClose: { tag: string; token: Tokens["HTML"] }[] = [];
23
26
 
24
27
  constructor(lexer: Lexer) {
25
28
  this.lexer = lexer;
@@ -77,8 +80,8 @@ export class Tokenizer {
77
80
  // remove trailing #s
78
81
  if (text.endsWith("#")) {
79
82
  const trimmed = rtrim(text, "#");
80
- if (!trimmed || trimmed.endsWith(" ")) {
81
- // CommonMark requires space before trailing #s
83
+ if (!trimmed || other.endingSpaceTabChar.test(trimmed)) {
84
+ // CommonMark requires a space or tab before trailing #s
82
85
  text = trimmed.trim();
83
86
  }
84
87
  }
@@ -167,13 +170,27 @@ export class Tokenizer {
167
170
  } else if (lastToken?.type === "blockquote") {
168
171
  // include continuation in nested blockquote
169
172
  const oldToken = lastToken;
170
- const newText = oldToken.raw + "\n" + lines.join("\n");
173
+ // The continuation lines belong to the same nesting frame as the
174
+ // nested blockquote, which already had one '>' marker stripped, so
175
+ // strip one marker from them too. Otherwise a restated marker after a
176
+ // lazy line is re-parsed as a spurious deeper blockquote.
177
+ const continuation = lines.join("\n");
178
+ const newText =
179
+ oldToken.raw + "\n" + continuation.replace(other.blockquoteSetextReplace2, "");
171
180
  // re-lexing the nested blockquote restarts at its first source line
172
181
  this.lexer.line = startLine + raw.split("\n").length - oldToken.raw.split("\n").length;
173
182
  const newToken = this.blockquote(newText)!;
174
183
  tokens[tokens.length - 1] = newToken;
175
184
 
176
- raw = raw.substring(0, raw.length - oldToken.raw.length) + newToken.raw;
185
+ // Only include continuation lines the nested blockquote actually
186
+ // consumed. Unconsumed trailing lines must stay out of `raw` so the
187
+ // lexer can still tokenize them (paragraph, next blockquote, etc.).
188
+ const leftover = newText.substring(newToken.raw.length).replace(/^\n/, "");
189
+ const leftoverCount = leftover ? leftover.split("\n").length : 0;
190
+ const consumed = leftoverCount ? lines.slice(0, -leftoverCount) : lines;
191
+ if (consumed.length > 0) {
192
+ raw = `${raw}\n${consumed.join("\n")}`;
193
+ }
177
194
  text = text.substring(0, text.length - oldToken.text.length) + newToken.text;
178
195
  break;
179
196
  } else if (lastToken?.type === "list") {
@@ -270,19 +287,23 @@ export class Tokenizer {
270
287
  raw = cap[0];
271
288
  src = src.substring(raw.length);
272
289
 
273
- let line = expandTabs(cap[2]!.split("\n", 1)[0]!, cap[1]!.length);
290
+ const firstLine = cap[2]!.split("\n", 1)[0]!;
291
+ const bulletIndent = cap[1]!.length;
292
+ let line = firstLine.replace(other.leadingSpaceTab, (whitespace) =>
293
+ expandTabs(whitespace, bulletIndent),
294
+ );
274
295
  let nextLine = src.split("\n", 1)[0] ?? "";
275
296
 
276
297
  let blankLine = !line.trim();
277
298
 
278
299
  let indent = 0;
279
300
  if (blankLine) {
280
- indent = cap[1]!.length + 1;
301
+ indent = bulletIndent + 1;
281
302
  } else {
282
303
  indent = line.search(other.nonSpaceChar); // Find first non-space char
283
304
  indent = indent > 4 ? 1 : indent; // Treat indented code blocks (> 4 spaces) as having only 1 indent
284
305
  itemContents = line.slice(indent);
285
- indent += cap[1]!.length;
306
+ indent += bulletIndent;
286
307
  }
287
308
 
288
309
  if (blankLine && other.blankLine.test(nextLine)) {
@@ -304,7 +325,9 @@ export class Tokenizer {
304
325
  while (src) {
305
326
  const rawLine = src.split("\n", 1)[0] ?? "";
306
327
  nextLine = rawLine;
307
- const nextLineWithoutTabs = nextLine.replace(other.tabCharGlobal, " ");
328
+ const nextLineWithoutTabs = nextLine.replace(other.leadingSpaceTab, (whitespace) =>
329
+ whitespace.replace(other.tabCharGlobal, " "),
330
+ );
308
331
 
309
332
  // End list item if found code fences
310
333
  if (fencesBeginRegex.test(nextLine)) {
@@ -416,25 +439,37 @@ export class Tokenizer {
416
439
  // save/restore top: blockTokens resets it to true on exit, and a nested list
417
440
  // lexed with a stale top=true would advance the sourcemap line counter
418
441
  const top = this.lexer.state.top;
442
+ // First pass: tokenize items and finalize list.loose from spacers before placing checkboxes
419
443
  for (const item of list.items) {
420
444
  this.lexer.state.top = false;
421
445
  item.tokens = this.lexer.blockTokens(item.text, []);
422
446
 
447
+ if (!list.loose) {
448
+ // Check if list should be loose
449
+ const spacers = item.tokens.filter((t) => t.type === "space");
450
+ const hasMultipleLineBreaks =
451
+ spacers.length > 0 && spacers.some((t) => other.anyLine.test(t.raw));
452
+
453
+ list.loose = hasMultipleLineBreaks;
454
+ }
455
+ }
456
+ this.lexer.state.top = top;
457
+
458
+ // Second pass: place task checkboxes using the final list.loose
459
+ for (const item of list.items) {
423
460
  const itemToken = item.tokens[0];
424
461
  if (item.task && (itemToken?.type === "text" || itemToken?.type === "paragraph")) {
425
462
  // Remove checkbox markdown from item tokens
426
463
  item.text = item.text.replace(other.listReplaceTask, "");
427
464
  itemToken.raw = itemToken.raw.replace(other.listReplaceTask, "");
428
465
  itemToken.text = itemToken.text.replace(other.listReplaceTask, "");
429
- for (let i = this.lexer.inlineQueue.length - 1; i >= 0; i--) {
430
- if (other.listIsTask.test(this.lexer.inlineQueue[i]!.src)) {
431
- this.lexer.inlineQueue[i]!.src = this.lexer.inlineQueue[i]!.src.replace(
432
- other.listReplaceTask,
433
- "",
434
- );
435
- break;
436
- }
437
- }
466
+ // DEVIATION: upstream strips the last queued src that looks like a task,
467
+ // which can be a later paragraph of the same item (`- [ ] a\n\n [ ] b`).
468
+ // The item token's own queue entry shares its tokens array.
469
+ const queued = this.lexer.inlineQueue.findLast(
470
+ (entry) => entry.tokens === itemToken.tokens,
471
+ );
472
+ if (queued) queued.src = queued.src.replace(other.listReplaceTask, "");
438
473
 
439
474
  const taskRaw = other.listTaskCheckbox.exec(item.raw);
440
475
  if (taskRaw) {
@@ -470,17 +505,7 @@ export class Tokenizer {
470
505
  } else if (item.task) {
471
506
  item.task = false;
472
507
  }
473
-
474
- if (!list.loose) {
475
- // Check if list should be loose
476
- const spacers = item.tokens.filter((t) => t.type === "space");
477
- const hasMultipleLineBreaks =
478
- spacers.length > 0 && spacers.some((t) => other.anyLine.test(t.raw));
479
-
480
- list.loose = hasMultipleLineBreaks;
481
- }
482
508
  }
483
- this.lexer.state.top = top;
484
509
 
485
510
  // Set all items to loose if list is loose
486
511
  if (list.loose) {
@@ -573,26 +598,26 @@ export class Tokenizer {
573
598
  * and update their sourceMap once they're closed.
574
599
  */
575
600
 
576
- const capEndsWith = (str?: string) => str && cap[0].trimEnd().endsWith(str);
577
-
578
- const tag = inline.tag.exec(src);
579
- const isHtmlClosed = capEndsWith(tag?.[0].slice(1));
580
-
581
- if (tag?.[0] && !isHtmlClosed) {
582
- // index where the token we just created will be inserted
583
- const tokenIdx = this.lexer.tokens.length;
584
- // first in last out
585
- this.pendingHtmlClose.unshift([tag[0], tokenIdx]);
586
- } else if (this.pendingHtmlClose.length) {
587
- for (const [pendingTag, index] of this.pendingHtmlClose) {
588
- if (capEndsWith(pendingTag.slice(1))) {
589
- const updateToken = this.lexer.tokens[index] as Tokens["HTML"];
590
- if (updateToken.sourceMap?.[1] && token.sourceMap?.[1]) {
591
- updateToken.sourceMap[1] = token.sourceMap[1];
592
- }
593
- this.pendingHtmlClose.shift();
594
- }
601
+ const openTag = other.htmlOpenTagName.exec(raw)?.[1]?.toLowerCase();
602
+ const closeTag = other.htmlEndingCloseTagName.exec(raw.trimEnd())?.[1]?.toLowerCase();
603
+
604
+ // a token ending in the closing tag of a pending html token closes it, even
605
+ // when it opens other tags itself (` <dt>a</dt>\n</dl>`); one that closes
606
+ // the tag it opens (`<div>a</div>`) is self-contained
607
+ const index =
608
+ closeTag === undefined || closeTag === openTag
609
+ ? -1
610
+ : this.pendingHtmlClose.findIndex((pending) => pending.tag === closeTag);
611
+ const pending = this.pendingHtmlClose[index];
612
+ if (pending) {
613
+ if (pending.token.sourceMap && token.sourceMap) {
614
+ pending.token.sourceMap[1] = token.sourceMap[1];
595
615
  }
616
+ // tags opened after the one just closed can no longer be closed
617
+ this.pendingHtmlClose.splice(0, index + 1);
618
+ } else if (openTag && openTag !== closeTag) {
619
+ // first in last out
620
+ this.pendingHtmlClose.unshift({ tag: openTag, token });
596
621
  }
597
622
 
598
623
  return token;
@@ -602,7 +627,7 @@ export class Tokenizer {
602
627
  const cap = block.def.exec(src);
603
628
  if (!cap) return undefined;
604
629
 
605
- const tag = cap[1]!.toLowerCase().replace(other.multipleSpaceGlobal, " ");
630
+ const tag = normalizeLabel(cap[1]!).replace(other.multipleSpaceGlobal, " ");
606
631
  const href = cap[2]
607
632
  ? cap[2].replace(other.hrefBrackets, "$1").replace(inline.anyPunctuation, "$1")
608
633
  : "";
@@ -687,7 +712,9 @@ export class Tokenizer {
687
712
  const cap = block.lheading.exec(src);
688
713
  if (!cap) return undefined;
689
714
 
690
- const text = cap[1]!.trim();
715
+ // DEVIATION: upstream keeps continuation lines' indentation; CommonMark (and
716
+ // GitHub, whose heading slugs would otherwise gain a dash per space) strips it
717
+ const text = cap[1]!.trim().replace(other.continuationIndent, "\n");
691
718
  const raw = rtrim(cap[0], "\n");
692
719
  return {
693
720
  type: "heading",
@@ -766,6 +793,10 @@ export class Tokenizer {
766
793
  const cap = inline.link.exec(src);
767
794
  if (!cap) return undefined;
768
795
 
796
+ if (isLabelEndInsideToken(src, cap[1]!, cap[0].startsWith("!") ? 2 : 1)) {
797
+ return;
798
+ }
799
+
769
800
  const trimmedUrl = cap[2]!.trim();
770
801
  if (trimmedUrl.startsWith("<")) {
771
802
  // commonmark requires matching angle brackets
@@ -818,8 +849,12 @@ export class Tokenizer {
818
849
  ): Tokens["Link"] | Tokens["Image"] | Tokens["Text"] | undefined {
819
850
  let cap;
820
851
  if ((cap = inline.reflink.exec(src)) ?? (cap = inline.nolink.exec(src))) {
852
+ if (isLabelEndInsideToken(src, cap[1]!, cap[0].startsWith("!") ? 2 : 1)) {
853
+ return;
854
+ }
855
+
821
856
  const linkStr = (cap[2] ?? cap[1])!.replace(/\s+/g, " ");
822
- const link = links[linkStr.toLowerCase()];
857
+ const link = links[normalizeLabel(linkStr)];
823
858
  if (!link) {
824
859
  const text = cap[0].charAt(0);
825
860
  return {
@@ -855,9 +890,12 @@ export class Tokenizer {
855
890
  delimTotal = lLength,
856
891
  midDelimTotal = 0;
857
892
 
858
- const endReg = match[0].startsWith("*")
859
- ? inline.emStrong.rDelimAst
860
- : inline.emStrong.rDelimUnd;
893
+ const delimChar = match[0][0];
894
+ // A mid-run opener (for example the second star of an unmatched `**`) must
895
+ // only pair with a delimiter that can only close, otherwise it steals the
896
+ // opener of a later span (`**a*b*c` must be `**a<em>b</em>c`).
897
+ const midRun = prevChar === delimChar;
898
+ const endReg = delimChar === "*" ? inline.emStrong.rDelimAst : inline.emStrong.rDelimUnd;
861
899
  endReg.lastIndex = 0;
862
900
 
863
901
  // Clip maskedSrc to same section of string as src (move to lexer?)
@@ -880,6 +918,11 @@ export class Tokenizer {
880
918
  midDelimTotal += rLength;
881
919
  continue; // CommonMark Emphasis Rules 9-10
882
920
  }
921
+ if (midRun) {
922
+ // A mid-run opener cannot close against an ambiguous delimiter that
923
+ // can also open; that delimiter opens its own emphasis span instead.
924
+ break;
925
+ }
883
926
  }
884
927
 
885
928
  delimTotal -= rLength;
@@ -1032,6 +1075,7 @@ export class Tokenizer {
1032
1075
  raw: cap[0],
1033
1076
  text,
1034
1077
  href,
1078
+ autolink: true,
1035
1079
  tokens: [
1036
1080
  {
1037
1081
  type: "text",
@@ -1069,6 +1113,7 @@ export class Tokenizer {
1069
1113
  raw: cap[0],
1070
1114
  text,
1071
1115
  href,
1116
+ autolink: true,
1072
1117
  tokens: [
1073
1118
  {
1074
1119
  type: "text",
@@ -1085,11 +1130,14 @@ export class Tokenizer {
1085
1130
  const cap = inline.text.exec(src);
1086
1131
  if (!cap) return undefined;
1087
1132
 
1133
+ const escaped = this.lexer.state.inRawBlock;
1088
1134
  return {
1089
1135
  type: "text",
1090
1136
  raw: cap[0],
1091
- text: cap[0],
1092
- escaped: this.lexer.state.inRawBlock,
1137
+ // raw HTML keeps whatever it was written with, everywhere else a numeric
1138
+ // character reference stands for the character itself
1139
+ text: escaped ? cap[0] : decodeNumericCharacterReferences(cap[0]),
1140
+ escaped,
1093
1141
  };
1094
1142
  }
1095
1143
 
package/src/types.ts CHANGED
@@ -141,6 +141,11 @@ export interface Tokens extends Record<string, BaseToken> {
141
141
  href: string;
142
142
  title: string | null;
143
143
  tokens: Token[];
144
+ /**
145
+ * Set for autolinks and extended (GFM) urls, where character references are
146
+ * not resolved, so the destination and text are literal.
147
+ */
148
+ autolink?: boolean;
144
149
  };
145
150
  Image: {
146
151
  type: "image";
package/src/utils.ts CHANGED
@@ -1,4 +1,5 @@
1
1
  import { type Lexer } from "./lexer.ts";
2
+ import { inline } from "./rules/inline.ts";
2
3
  import { other } from "./rules/other.ts";
3
4
  import { type HTMLAttrs, type PantsdownConfig, type SourceMap, type Tokens } from "./types.ts";
4
5
 
@@ -154,6 +155,24 @@ export function fixLocalImageHref(href: string, config: PantsdownConfig): string
154
155
  }
155
156
  }
156
157
 
158
+ /**
159
+ * Numeric character references are recognized outside code and are equivalent to the
160
+ * character they name. Values that are zero, out of range, or a surrogate become the
161
+ * replacement character.
162
+ */
163
+ export function decodeNumericCharacterReferences(text: string) {
164
+ return text.replace(
165
+ other.numericCharacterReference,
166
+ (_, dec: string | undefined, hex: string | undefined) => {
167
+ const code = dec === undefined ? Number.parseInt(hex!, 16) : Number.parseInt(dec, 10);
168
+ if (code === 0 || code > 0x10ffff || (code >= 0xd800 && code <= 0xdfff)) {
169
+ return "\uFFFD";
170
+ }
171
+ return String.fromCodePoint(code);
172
+ },
173
+ );
174
+ }
175
+
157
176
  export function cleanUrl(href: string) {
158
177
  try {
159
178
  href = encodeURI(href).replace(other.percentDecode, "%");
@@ -252,6 +271,20 @@ export function trimTrailingBlankLines(str: string) {
252
271
  return lines.slice(0, end + 1).join("\n");
253
272
  }
254
273
 
274
+ /**
275
+ * Normalizes a link label so definitions and references can be matched.
276
+ * CommonMark asks for a Unicode case fold, which `toLowerCase()` does not
277
+ * reach: `ẞ` lowercases to `ß` and so never meets `SS`. Round-tripping
278
+ * through upper case does, as in commonmark.js; the final `toLowerCase()`
279
+ * keeps the folded label lower case, the form `def.tag` has always used.
280
+ */
281
+ export function normalizeLabel(label: string) {
282
+ // The spec also asks for leading and trailing spaces, tabs and line endings
283
+ // to be stripped. Doing it here keeps every call site in agreement: the
284
+ // definition and the reference have to normalize to the same key.
285
+ return label.trim().toLowerCase().toUpperCase().toLowerCase();
286
+ }
287
+
255
288
  export function findClosingBracket(str: string, b: string) {
256
289
  if (!b[1] || !str.includes(b[1])) {
257
290
  return -1;
@@ -281,22 +314,43 @@ export function outputLink(
281
314
  link: Pick<Tokens["Link"], "href" | "title">,
282
315
  raw: string,
283
316
  lexer: Lexer,
284
- ): Tokens["Link"] | Tokens["Image"] {
317
+ ): Tokens["Link"] | Tokens["Image"] | undefined {
285
318
  const href = link.href;
286
319
  const title = link.title || null;
287
320
  const text = cap[1]?.replace(other.outputLinkReplace, "$1") ?? "";
321
+ const isImage = cap[0]?.charAt(0) === "!";
288
322
 
289
323
  lexer.state.inLink = true;
290
- const token: Tokens["Link"] | Tokens["Image"] = {
291
- type: cap[0]?.charAt(0) === "!" ? "image" : "link",
324
+ const outerLinkEmitted = lexer.state.linkEmitted;
325
+ const outerInRawBlock = lexer.state.inRawBlock;
326
+ lexer.state.linkEmitted = false;
327
+ const tokens = lexer.inlineTokens(text);
328
+ // widen: TS keeps the `= false` narrowing across the inlineTokens call that sets it
329
+ const textHasLink = lexer.state.linkEmitted as boolean;
330
+ lexer.state.linkEmitted = outerLinkEmitted;
331
+ lexer.state.inLink = false;
332
+
333
+ if (!isImage) {
334
+ // CommonMark: "Links may not contain other links, at any level of nesting."
335
+ // Bail so the caller falls through to text and the inner link is the one kept.
336
+ // Images are exempt: their text is flattened into an alt attribute.
337
+ if (textHasLink) {
338
+ // these tokens are discarded, so undo the raw-block state they opened;
339
+ // leaving it set would suppress escaping for the text that is re-scanned
340
+ lexer.state.inRawBlock = outerInRawBlock;
341
+ return;
342
+ }
343
+ lexer.state.linkEmitted = true;
344
+ }
345
+
346
+ return {
347
+ type: isImage ? "image" : "link",
292
348
  raw,
293
349
  href,
294
350
  title,
295
351
  text,
296
- tokens: lexer.inlineTokens(text),
352
+ tokens,
297
353
  };
298
- lexer.state.inLink = false;
299
- return token;
300
354
  }
301
355
 
302
356
  export function indentCodeCompensation(raw: string, text: string) {
@@ -318,15 +372,55 @@ export function indentCodeCompensation(raw: string, text: string) {
318
372
 
319
373
  const [indentInNode] = matchIndentInNode;
320
374
 
321
- if (indentToCode && indentInNode.length >= indentToCode.length) {
322
- return node.slice(indentToCode.length);
323
- }
324
-
325
- return node;
375
+ // Up to the fence's own indentation is removed from each line, so a line
376
+ // indented less than the fence loses whatever indentation it has.
377
+ return node.slice(Math.min(indentInNode.length, indentToCode?.length ?? 0));
326
378
  })
327
379
  .join("\n");
328
380
  }
329
381
 
382
+ /**
383
+ * Does the link label end inside a raw token (html tag or autolink) that starts
384
+ * within it? Such a token takes precedence over the link, so its closing `]`
385
+ * does not close the label.
386
+ */
387
+ export function isLabelEndInsideToken(src: string, label: string, labelStart: number) {
388
+ if (!label.includes("<")) {
389
+ return false;
390
+ }
391
+
392
+ for (let i = 0; i < label.length; i++) {
393
+ if (label[i] === "\\") {
394
+ i++;
395
+ continue;
396
+ }
397
+
398
+ if (label[i] === "`") {
399
+ const code = inline.code.exec(label.slice(i));
400
+ if (code) {
401
+ i += code[0].length - 1;
402
+ continue;
403
+ }
404
+ }
405
+
406
+ if (label[i] !== "<") {
407
+ continue;
408
+ }
409
+
410
+ const tokenSrc = src.slice(labelStart + i);
411
+ const token = inline.tag.exec(tokenSrc) ?? inline.autolink.exec(tokenSrc);
412
+ if (!token) {
413
+ continue;
414
+ }
415
+
416
+ if (token[0].length > label.length - i) {
417
+ return true;
418
+ }
419
+ i += token[0].length - 1;
420
+ }
421
+ return false;
422
+ }
423
+
330
424
  function makeAlertRegex(type: string) {
331
425
  return new RegExp(`^(?:\\[\\!${type.toUpperCase()}\\]|[\\*]{2}${type}[\\*]{2})[ \\t]*\\n?`);
332
426
  }