pantsdown 2.2.3 → 2.2.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/tokenizer.ts CHANGED
@@ -3,953 +3,945 @@ import { block } from "./rules/block.ts";
3
3
  import { inline } from "./rules/inline.ts";
4
4
  import { type Links, type Tokens } from "./types.ts";
5
5
  import {
6
- ALERTS,
7
- escape,
8
- findClosingBracket,
9
- indentCodeCompensation,
10
- outputLink,
11
- rtrim,
12
- splitCells,
6
+ ALERTS,
7
+ escape,
8
+ findClosingBracket,
9
+ indentCodeCompensation,
10
+ outputLink,
11
+ rtrim,
12
+ splitCells,
13
13
  } from "./utils.ts";
14
14
 
15
15
  /**
16
16
  * The tokenizer defines how to turn markdown text into tokens.
17
17
  */
18
18
  export class Tokenizer {
19
- private lexer: Lexer;
20
- pendingHtmlClose: [tag: string, index: number][] = [];
21
-
22
- constructor(lexer: Lexer) {
23
- this.lexer = lexer;
24
- }
25
-
26
- space(src: string): Tokens["Space"] | undefined {
27
- const cap = block.newline.exec(src);
28
- if (cap && cap[0].length > 0) {
29
- // call getSourceMap to increment "this.lexer.line"
30
- this.lexer.getSourceMap(cap[0]);
31
- return {
32
- type: "space",
33
- raw: cap[0],
34
- };
35
- }
36
- return;
37
- }
38
-
39
- code(src: string): Tokens["Code"] | undefined {
40
- const cap = block.code.exec(src);
41
- if (!cap) return undefined;
42
-
43
- const text = cap[0].replace(/^ {1,4}/gm, "");
44
- return {
45
- type: "code",
46
- raw: cap[0],
47
- codeBlockStyle: "indented",
48
- text: rtrim(text, "\n"),
49
- sourceMap: this.lexer.getSourceMap(cap[0]),
50
- };
51
- }
52
-
53
- fences(src: string): Tokens["Code"] | undefined {
54
- const cap = block.fences.exec(src);
55
- if (!cap) return undefined;
56
-
57
- const raw = cap[0];
58
- const text = indentCodeCompensation(raw, cap[3] ?? "");
59
-
60
- return {
61
- type: "code",
62
- raw,
63
- lang: cap[2] ? cap[2].trim().replace(inline.anyPunctuation, "$1") : cap[2],
64
- text,
65
- sourceMap: this.lexer.getSourceMap(raw),
66
- };
67
- }
68
-
69
- heading(src: string): Tokens["Heading"] | undefined {
70
- const cap = block.heading.exec(src);
71
- if (!cap) return undefined;
72
- let text = cap[2]!.trim();
73
-
74
- // remove trailing #s
75
- if (text.endsWith("#")) {
76
- const trimmed = rtrim(text, "#");
77
- if (!trimmed || trimmed.endsWith(" ")) {
78
- // CommonMark requires space before trailing #s
79
- text = trimmed.trim();
80
- }
81
- }
82
-
83
- return {
84
- type: "heading",
19
+ private lexer: Lexer;
20
+ pendingHtmlClose: [tag: string, index: number][] = [];
21
+
22
+ constructor(lexer: Lexer) {
23
+ this.lexer = lexer;
24
+ }
25
+
26
+ space(src: string): Tokens["Space"] | undefined {
27
+ const cap = block.newline.exec(src);
28
+ if (cap && cap[0].length > 0) {
29
+ // call getSourceMap to increment "this.lexer.line"
30
+ this.lexer.getSourceMap(cap[0]);
31
+ return {
32
+ type: "space",
85
33
  raw: cap[0],
86
- depth: cap[1]!.length,
87
- text,
88
- tokens: this.lexer.inline(text),
89
- sourceMap: this.lexer.getSourceMap(cap[0]),
90
- };
91
- }
92
-
93
- hr(src: string): Tokens["Hr"] | undefined {
94
- const cap = block.hr.exec(src);
95
- if (!cap) return undefined;
96
-
97
- return {
98
- type: "hr",
99
- raw: cap[0],
100
- sourceMap: this.lexer.getSourceMap(cap[0]),
101
- };
102
- }
103
-
104
- blockquote(src: string): Tokens["Blockquote"] | Tokens["Alert"] | undefined {
105
- const cap = block.blockquote.exec(src);
106
- if (!cap) return undefined;
107
-
108
- // precede setext continuation with 4 spaces so it isn't a setext
109
- let text = cap[0].replace(/\n {0,3}((?:=+|-+) *)(?=\n|$)/g, "\n $1");
110
- text = rtrim(text.replace(/^ *>[ \t]?/gm, ""), "\n");
111
- const top = this.lexer.state.top;
112
- this.lexer.state.top = true;
113
- const tokens = this.lexer.blockTokens(text, []);
114
- this.lexer.state.top = top;
115
- this.lexer.line++;
116
-
117
- const blockquoteToken: Tokens["Blockquote"] = {
118
- type: "blockquote",
119
- raw: cap[0],
120
- tokens,
121
- text,
122
- };
123
-
124
- const matchedAlertVariant = ALERTS.find(({ regex }) => regex.test(blockquoteToken.text));
125
-
126
- if (matchedAlertVariant) {
127
- const { variant, icon, regex } = matchedAlertVariant;
128
-
129
- const firstToken = blockquoteToken.tokens[0] as Tokens["Paragraph"];
130
-
131
- text = firstToken.raw.replace(regex, "");
132
-
133
- // since we're modifying firstToken.raw by removing the first line,
134
- // resulting tokens also change and thus also the sourceMap that
135
- // was generated for the blockquote
136
- if (firstToken.sourceMap?.[0]) firstToken.sourceMap[0]++;
137
-
138
- firstToken.tokens = this.lexer.inlineTokens(text, []);
139
-
140
- const alertToken: Tokens["Alert"] = {
141
- ...blockquoteToken,
142
- type: "alert",
143
- variant: variant as Tokens["Alert"]["variant"],
144
- icon,
145
- };
146
-
147
- return alertToken;
148
- }
149
-
150
- return blockquoteToken;
151
- }
152
-
153
- list(src: string): Tokens["List"] | undefined {
154
- let cap = block.list.exec(src);
155
- if (!cap) return undefined;
156
-
157
- let bull = cap[1]!.trim();
158
- const isordered = bull.length > 1;
159
-
160
- const list: Tokens["List"] = {
161
- type: "list",
162
- raw: "",
163
- ordered: isordered,
164
- start: isordered ? +bull.slice(0, -1) : "",
165
- loose: false,
166
- items: [] as Tokens["ListItem"][],
167
- };
168
-
169
- bull = isordered ? `\\d{1,9}\\${bull.slice(-1)}` : `\\${bull}`;
170
-
171
- // Get next list item
172
- const itemRegex = new RegExp(`^( {0,3}${bull})((?:[\t ][^\\n]*)?(?:\\n|$))`);
173
- let raw = "";
174
- let itemContents = "";
175
- let endsWithBlankLine = false;
176
- // Check if current bullet point can start a new List Item
177
- while (src) {
178
- let endEarly = false;
179
- if (!(cap = itemRegex.exec(src))) {
180
- break;
181
- }
182
-
183
- if (block.hr.test(src)) {
184
- // End list if bullet was actually HR (possibly move into itemRegex?)
185
- break;
186
- }
187
-
188
- raw = cap[0];
189
- src = src.substring(raw.length);
190
-
191
- let line = cap[2]!
192
- .split("\n", 1)[0]!
193
- .replace(/^\t+/, (t: string) => " ".repeat(3 * t.length));
194
- let nextLine = src.split("\n", 1)[0] ?? "";
195
-
196
- let indent = 0;
197
- indent = cap[2]!.search(/[^ ]/); // Find first non-space char
198
- indent = indent > 4 ? 1 : indent; // Treat indented code blocks (> 4 spaces) as having only 1 indent
199
- itemContents = line.slice(indent);
200
- indent += cap[1]!.length;
201
-
202
- let blankLine = false;
203
-
204
- if (!line && /^ *$/.test(nextLine)) {
205
- // Items begin with at most one blank line
206
- raw += nextLine + "\n";
207
- src = src.substring(nextLine.length + 1);
208
- endEarly = true;
209
- }
210
-
211
- if (!endEarly) {
212
- const nextBulletRegex = new RegExp(
213
- `^ {0,${Math.min(
214
- 3,
215
- indent - 1,
216
- )}}(?:[*+-]|\\d{1,9}[.)])((?:[ \t][^\\n]*)?(?:\\n|$))`,
217
- );
218
- const hrRegex = new RegExp(
219
- `^ {0,${Math.min(
220
- 3,
221
- indent - 1,
222
- )}}((?:- *){3,}|(?:_ *){3,}|(?:\\* *){3,})(?:\\n+|$)`,
223
- );
224
- const fencesBeginRegex = new RegExp(
225
- `^ {0,${Math.min(3, indent - 1)}}(?:\`\`\`|~~~)`,
226
- );
227
- const headingBeginRegex = new RegExp(`^ {0,${Math.min(3, indent - 1)}}#`);
228
-
229
- // Check if following lines should be included in List Item
230
- while (src) {
231
- const rawLine = src.split("\n", 1)[0] ?? "";
232
- nextLine = rawLine;
233
-
234
- // End list item if found code fences
235
- if (fencesBeginRegex.test(nextLine)) {
236
- break;
237
- }
238
-
239
- // End list item if found start of new heading
240
- if (headingBeginRegex.test(nextLine)) {
241
- break;
242
- }
243
-
244
- // End list item if found start of new bullet
245
- if (nextBulletRegex.test(nextLine)) {
246
- break;
247
- }
248
-
249
- // Horizontal rule found
250
- if (hrRegex.test(src)) {
251
- break;
252
- }
253
-
254
- if (nextLine.search(/[^ ]/) >= indent || !nextLine.trim()) {
255
- // Dedent if possible
256
- itemContents += "\n" + nextLine.slice(indent);
257
- } else {
258
- // not enough indentation
259
- if (blankLine) {
260
- break;
261
- }
262
-
263
- // paragraph continuation unless last line was a different block level element
264
- if (line.search(/[^ ]/) >= 4) {
265
- // indented code block
266
- break;
267
- }
268
- if (fencesBeginRegex.test(line)) {
269
- break;
270
- }
271
- if (headingBeginRegex.test(line)) {
272
- break;
273
- }
274
- if (hrRegex.test(line)) {
275
- break;
276
- }
277
-
278
- itemContents += "\n" + nextLine;
279
- }
280
-
281
- if (!blankLine && !nextLine.trim()) {
282
- // Check if current line is blank
283
- blankLine = true;
284
- }
285
-
286
- raw += rawLine + "\n";
287
- src = src.substring(rawLine.length + 1);
288
- line = nextLine.slice(indent);
289
- }
290
- }
291
-
292
- if (!list.loose) {
293
- // If the previous item ended with a blank line, the list is loose
294
- if (endsWithBlankLine) {
295
- list.loose = true;
296
- } else if (/\n *\n *$/.test(raw)) {
297
- endsWithBlankLine = true;
298
- }
299
- }
300
-
301
- let istask: RegExpExecArray | null = null;
302
- let ischecked: boolean | undefined;
303
- // Check for task list items
304
- istask = /^\[[ xX]\] /.exec(itemContents);
305
- if (istask) {
306
- ischecked = istask[0] !== "[ ] ";
307
- itemContents = itemContents.replace(/^\[[ xX]\] +/, "");
34
+ };
35
+ }
36
+ return;
37
+ }
38
+
39
+ code(src: string): Tokens["Code"] | undefined {
40
+ const cap = block.code.exec(src);
41
+ if (!cap) return undefined;
42
+
43
+ const text = cap[0].replace(/^ {1,4}/gm, "");
44
+ return {
45
+ type: "code",
46
+ raw: cap[0],
47
+ codeBlockStyle: "indented",
48
+ text: rtrim(text, "\n"),
49
+ sourceMap: this.lexer.getSourceMap(cap[0]),
50
+ };
51
+ }
52
+
53
+ fences(src: string): Tokens["Code"] | undefined {
54
+ const cap = block.fences.exec(src);
55
+ if (!cap) return undefined;
56
+
57
+ const raw = cap[0];
58
+ const text = indentCodeCompensation(raw, cap[3] ?? "");
59
+
60
+ return {
61
+ type: "code",
62
+ raw,
63
+ lang: cap[2] ? cap[2].trim().replace(inline.anyPunctuation, "$1") : cap[2],
64
+ text,
65
+ sourceMap: this.lexer.getSourceMap(raw),
66
+ };
67
+ }
68
+
69
+ heading(src: string): Tokens["Heading"] | undefined {
70
+ const cap = block.heading.exec(src);
71
+ if (!cap) return undefined;
72
+ let text = cap[2]!.trim();
73
+
74
+ // remove trailing #s
75
+ if (text.endsWith("#")) {
76
+ const trimmed = rtrim(text, "#");
77
+ if (!trimmed || trimmed.endsWith(" ")) {
78
+ // CommonMark requires space before trailing #s
79
+ text = trimmed.trim();
80
+ }
81
+ }
82
+
83
+ return {
84
+ type: "heading",
85
+ raw: cap[0],
86
+ depth: cap[1]!.length,
87
+ text,
88
+ tokens: this.lexer.inline(text),
89
+ sourceMap: this.lexer.getSourceMap(cap[0]),
90
+ };
91
+ }
92
+
93
+ hr(src: string): Tokens["Hr"] | undefined {
94
+ const cap = block.hr.exec(src);
95
+ if (!cap) return undefined;
96
+
97
+ return {
98
+ type: "hr",
99
+ raw: cap[0],
100
+ sourceMap: this.lexer.getSourceMap(cap[0]),
101
+ };
102
+ }
103
+
104
+ blockquote(src: string): Tokens["Blockquote"] | Tokens["Alert"] | undefined {
105
+ const cap = block.blockquote.exec(src);
106
+ if (!cap) return undefined;
107
+
108
+ // precede setext continuation with 4 spaces so it isn't a setext
109
+ let text = cap[0].replace(/\n {0,3}((?:=+|-+) *)(?=\n|$)/g, "\n $1");
110
+ text = rtrim(text.replace(/^ *>[ \t]?/gm, ""), "\n");
111
+ const top = this.lexer.state.top;
112
+ this.lexer.state.top = true;
113
+ const tokens = this.lexer.blockTokens(text, []);
114
+ this.lexer.state.top = top;
115
+ this.lexer.line++;
116
+
117
+ const blockquoteToken: Tokens["Blockquote"] = {
118
+ type: "blockquote",
119
+ raw: cap[0],
120
+ tokens,
121
+ text,
122
+ };
123
+
124
+ const matchedAlertVariant = ALERTS.find(({ regex }) => regex.test(blockquoteToken.text));
125
+
126
+ if (matchedAlertVariant) {
127
+ const { variant, icon, regex } = matchedAlertVariant;
128
+
129
+ const firstToken = blockquoteToken.tokens[0] as Tokens["Paragraph"];
130
+
131
+ text = firstToken.raw.replace(regex, "");
132
+
133
+ // since we're modifying firstToken.raw by removing the first line,
134
+ // resulting tokens also change and thus also the sourceMap that
135
+ // was generated for the blockquote
136
+ if (firstToken.sourceMap?.[0]) firstToken.sourceMap[0]++;
137
+
138
+ firstToken.tokens = this.lexer.inlineTokens(text, []);
139
+
140
+ const alertToken: Tokens["Alert"] = {
141
+ ...blockquoteToken,
142
+ type: "alert",
143
+ variant: variant as Tokens["Alert"]["variant"],
144
+ icon,
145
+ };
146
+
147
+ return alertToken;
148
+ }
149
+
150
+ return blockquoteToken;
151
+ }
152
+
153
+ list(src: string): Tokens["List"] | undefined {
154
+ let cap = block.list.exec(src);
155
+ if (!cap) return undefined;
156
+
157
+ let bull = cap[1]!.trim();
158
+ const isordered = bull.length > 1;
159
+
160
+ const list: Tokens["List"] = {
161
+ type: "list",
162
+ raw: "",
163
+ ordered: isordered,
164
+ start: isordered ? +bull.slice(0, -1) : "",
165
+ loose: false,
166
+ items: [] as Tokens["ListItem"][],
167
+ };
168
+
169
+ bull = isordered ? `\\d{1,9}\\${bull.slice(-1)}` : `\\${bull}`;
170
+
171
+ // Get next list item
172
+ const itemRegex = new RegExp(`^( {0,3}${bull})((?:[\t ][^\\n]*)?(?:\\n|$))`);
173
+ let raw = "";
174
+ let itemContents = "";
175
+ let endsWithBlankLine = false;
176
+ // Check if current bullet point can start a new List Item
177
+ while (src) {
178
+ let endEarly = false;
179
+ if (!(cap = itemRegex.exec(src))) {
180
+ break;
181
+ }
182
+
183
+ if (block.hr.test(src)) {
184
+ // End list if bullet was actually HR (possibly move into itemRegex?)
185
+ break;
186
+ }
187
+
188
+ raw = cap[0];
189
+ src = src.substring(raw.length);
190
+
191
+ let line = cap[2]!
192
+ .split("\n", 1)[0]!
193
+ .replace(/^\t+/, (t: string) => " ".repeat(3 * t.length));
194
+ let nextLine = src.split("\n", 1)[0] ?? "";
195
+
196
+ let indent = 0;
197
+ indent = cap[2]!.search(/[^ ]/); // Find first non-space char
198
+ indent = indent > 4 ? 1 : indent; // Treat indented code blocks (> 4 spaces) as having only 1 indent
199
+ itemContents = line.slice(indent);
200
+ indent += cap[1]!.length;
201
+
202
+ let blankLine = false;
203
+
204
+ if (!line && /^ *$/.test(nextLine)) {
205
+ // Items begin with at most one blank line
206
+ raw += nextLine + "\n";
207
+ src = src.substring(nextLine.length + 1);
208
+ endEarly = true;
209
+ }
210
+
211
+ if (!endEarly) {
212
+ const nextBulletRegex = new RegExp(
213
+ `^ {0,${Math.min(3, indent - 1)}}(?:[*+-]|\\d{1,9}[.)])((?:[ \t][^\\n]*)?(?:\\n|$))`,
214
+ );
215
+ const hrRegex = new RegExp(
216
+ `^ {0,${Math.min(3, indent - 1)}}((?:- *){3,}|(?:_ *){3,}|(?:\\* *){3,})(?:\\n+|$)`,
217
+ );
218
+ const fencesBeginRegex = new RegExp(`^ {0,${Math.min(3, indent - 1)}}(?:\`\`\`|~~~)`);
219
+ const headingBeginRegex = new RegExp(`^ {0,${Math.min(3, indent - 1)}}#`);
220
+
221
+ // Check if following lines should be included in List Item
222
+ while (src) {
223
+ const rawLine = src.split("\n", 1)[0] ?? "";
224
+ nextLine = rawLine;
225
+
226
+ // End list item if found code fences
227
+ if (fencesBeginRegex.test(nextLine)) {
228
+ break;
229
+ }
230
+
231
+ // End list item if found start of new heading
232
+ if (headingBeginRegex.test(nextLine)) {
233
+ break;
234
+ }
235
+
236
+ // End list item if found start of new bullet
237
+ if (nextBulletRegex.test(nextLine)) {
238
+ break;
239
+ }
240
+
241
+ // Horizontal rule found
242
+ if (hrRegex.test(src)) {
243
+ break;
244
+ }
245
+
246
+ if (nextLine.search(/[^ ]/) >= indent || !nextLine.trim()) {
247
+ // Dedent if possible
248
+ itemContents += "\n" + nextLine.slice(indent);
249
+ } else {
250
+ // not enough indentation
251
+ if (blankLine) {
252
+ break;
253
+ }
254
+
255
+ // paragraph continuation unless last line was a different block level element
256
+ if (line.search(/[^ ]/) >= 4) {
257
+ // indented code block
258
+ break;
259
+ }
260
+ if (fencesBeginRegex.test(line)) {
261
+ break;
262
+ }
263
+ if (headingBeginRegex.test(line)) {
264
+ break;
265
+ }
266
+ if (hrRegex.test(line)) {
267
+ break;
268
+ }
269
+
270
+ itemContents += "\n" + nextLine;
271
+ }
272
+
273
+ if (!blankLine && !nextLine.trim()) {
274
+ // Check if current line is blank
275
+ blankLine = true;
276
+ }
277
+
278
+ raw += rawLine + "\n";
279
+ src = src.substring(rawLine.length + 1);
280
+ line = nextLine.slice(indent);
308
281
  }
309
-
310
- list.items.push({
311
- type: "list_item",
312
- raw,
313
- task: Boolean(istask),
314
- checked: ischecked,
315
- loose: false,
316
- text: itemContents,
317
- tokens: [],
318
- sourceMap: this.lexer.getSourceMap(raw),
319
- });
320
-
321
- list.raw += raw;
322
- }
323
-
324
- // Do not consume newlines at end of final item. Alternatively, make itemRegex *start* with any newlines to simplify/speed up endsWithBlankLine logic
325
- const lastTrimmed = raw.trimEnd();
326
-
327
- if (list.items[list.items.length - 1]!.sourceMap) {
328
- this.lexer.line -= raw.length - lastTrimmed.length;
329
- }
330
-
331
- list.items[list.items.length - 1]!.raw = lastTrimmed;
332
- list.items[list.items.length - 1]!.text = itemContents.trimEnd();
333
- list.raw = list.raw.trimEnd();
334
-
335
- // Item child tokens handled here at end because we needed to have the final item to trim it first
336
- for (let i = 0, listItemsLen = list.items.length; i < listItemsLen; i++) {
337
- this.lexer.state.top = false;
338
- list.items[i]!.tokens = this.lexer.blockTokens(list.items[i]!.text, []);
339
-
340
- if (!list.loose) {
341
- // Check if list should be loose
342
- const spacers = list.items[i]!.tokens.filter((t) => t.type === "space");
343
- const hasMultipleLineBreaks =
344
- // eslint-disable-next-line
345
- spacers.length > 0 && spacers.some((t: any) => /\n.*\n/.test(t.raw));
346
-
347
- list.loose = hasMultipleLineBreaks;
282
+ }
283
+
284
+ if (!list.loose) {
285
+ // If the previous item ended with a blank line, the list is loose
286
+ if (endsWithBlankLine) {
287
+ list.loose = true;
288
+ } else if (/\n *\n *$/.test(raw)) {
289
+ endsWithBlankLine = true;
348
290
  }
349
- }
350
-
351
- // Set all items to loose if list is loose
352
- if (list.loose) {
353
- for (let i = 0, listItemsLen = list.items.length; i < listItemsLen; i++) {
354
- list.items[i]!.loose = true;
291
+ }
292
+
293
+ let istask: RegExpExecArray | null = null;
294
+ let ischecked: boolean | undefined;
295
+ // Check for task list items
296
+ istask = /^\[[ xX]\] /.exec(itemContents);
297
+ if (istask) {
298
+ ischecked = istask[0] !== "[ ] ";
299
+ itemContents = itemContents.replace(/^\[[ xX]\] +/, "");
300
+ }
301
+
302
+ list.items.push({
303
+ type: "list_item",
304
+ raw,
305
+ task: Boolean(istask),
306
+ checked: ischecked,
307
+ loose: false,
308
+ text: itemContents,
309
+ tokens: [],
310
+ sourceMap: this.lexer.getSourceMap(raw),
311
+ });
312
+
313
+ list.raw += raw;
314
+ }
315
+
316
+ // Do not consume newlines at end of final item. Alternatively, make itemRegex *start* with any newlines to simplify/speed up endsWithBlankLine logic
317
+ const lastTrimmed = raw.trimEnd();
318
+
319
+ if (list.items[list.items.length - 1]!.sourceMap) {
320
+ this.lexer.line -= raw.length - lastTrimmed.length;
321
+ }
322
+
323
+ list.items[list.items.length - 1]!.raw = lastTrimmed;
324
+ list.items[list.items.length - 1]!.text = itemContents.trimEnd();
325
+ list.raw = list.raw.trimEnd();
326
+
327
+ // Item child tokens handled here at end because we needed to have the final item to trim it first
328
+ for (let i = 0, listItemsLen = list.items.length; i < listItemsLen; i++) {
329
+ this.lexer.state.top = false;
330
+ list.items[i]!.tokens = this.lexer.blockTokens(list.items[i]!.text, []);
331
+
332
+ if (!list.loose) {
333
+ // Check if list should be loose
334
+ const spacers = list.items[i]!.tokens.filter((t) => t.type === "space");
335
+ const hasMultipleLineBreaks =
336
+ // eslint-disable-next-line
337
+ spacers.length > 0 && spacers.some((t: any) => /\n.*\n/.test(t.raw));
338
+
339
+ list.loose = hasMultipleLineBreaks;
340
+ }
341
+ }
342
+
343
+ // Set all items to loose if list is loose
344
+ if (list.loose) {
345
+ for (let i = 0, listItemsLen = list.items.length; i < listItemsLen; i++) {
346
+ list.items[i]!.loose = true;
347
+ }
348
+ }
349
+
350
+ return list;
351
+ }
352
+
353
+ footnote(src: string): Tokens["Footnote"] | undefined {
354
+ const cap = block.footnote.exec(src);
355
+ if (!cap) return undefined;
356
+
357
+ const label = cap[1] ?? "";
358
+ let text = cap[2] ? rtrim(cap[2].replace(/^ *[ \t]?/gm, ""), "\n") : "";
359
+
360
+ text += `<a href="#footnote-ref-${encodeURIComponent(
361
+ label,
362
+ )}" data-footnote-backref aria-label="Back to reference ${label}"> ↩</a>`;
363
+
364
+ this.lexer.state.top = false;
365
+ const tokens = this.lexer.blockTokens(text, []);
366
+
367
+ const token: Tokens["Footnote"] = {
368
+ type: "footnote",
369
+ raw: cap[0],
370
+ text: text,
371
+ label: label,
372
+ content: tokens,
373
+ sourceMap: this.lexer.getSourceMap(cap[0]),
374
+ };
375
+
376
+ return token;
377
+ }
378
+
379
+ html(src: string): Tokens["HTML"] | undefined {
380
+ const cap = block.html.exec(src);
381
+ if (!cap) return undefined;
382
+
383
+ const token: Tokens["HTML"] = {
384
+ type: "html",
385
+ block: true,
386
+ raw: cap[0],
387
+ pre: cap[1] === "pre" || cap[1] === "script" || cap[1] === "style",
388
+ text: cap[0],
389
+ sourceMap: this.lexer.getSourceMap(cap[0]),
390
+ };
391
+
392
+ /*
393
+ * Sometimes an html token does not contain its closing tag.
394
+ *
395
+ * The following markdown:
396
+ *
397
+ * 130 <details>
398
+ * 131 <summary><h4>Hello World</h4></summary>
399
+ * 132
400
+ * 133 Here some non-html markdown
401
+ * 134 </details>
402
+ *
403
+ * would result in 3 tokens:
404
+ * [
405
+ * {
406
+ * type: "html",
407
+ * raw: "<details>↵ <summary><h4>PUT</h4></summary>↵↵",
408
+ * sourceMap: [130, 131]
409
+ * },
410
+ * {
411
+ * type: "paragraph",
412
+ * raw: "Here some non-html markdown",
413
+ * sourceMap: [133, 133]
414
+ * },
415
+ * {
416
+ * type: "html",
417
+ * raw: "</details>",
418
+ * }
419
+ * ]
420
+ *
421
+ * sourceMap metadata in first token will be incorrect, because it will
422
+ * not take its children into consideration when it should.
423
+ *
424
+ * To solve that we keep track of html tokens that are pending to be closed
425
+ * and update their sourceMap once they're closed.
426
+ */
427
+
428
+ const capEndsWith = (str?: string) => str && cap[0].trimEnd().endsWith(str);
429
+
430
+ const tag = inline.tag.exec(src);
431
+ const isHtmlClosed = capEndsWith(tag?.[0].slice(1));
432
+
433
+ if (tag?.[0] && !isHtmlClosed) {
434
+ // index where the token we just created will be inserted
435
+ const tokenIdx = this.lexer.tokens.length;
436
+ // first in last out
437
+ this.pendingHtmlClose.unshift([tag[0], tokenIdx]);
438
+ } else if (this.pendingHtmlClose.length) {
439
+ for (const [pendingTag, index] of this.pendingHtmlClose) {
440
+ if (capEndsWith(pendingTag.slice(1))) {
441
+ const updateToken = this.lexer.tokens[index] as Tokens["HTML"];
442
+ if (updateToken.sourceMap?.[1] && token.sourceMap?.[1]) {
443
+ updateToken.sourceMap[1] = token.sourceMap[1];
444
+ }
445
+ this.pendingHtmlClose.shift();
355
446
  }
356
- }
357
-
358
- return list;
359
- }
360
-
361
- footnote(src: string): Tokens["Footnote"] | undefined {
362
- const cap = block.footnote.exec(src);
363
- if (!cap) return undefined;
364
-
365
- const label = cap[1] ?? "";
366
- let text = cap[2] ? rtrim(cap[2].replace(/^ *[ \t]?/gm, ""), "\n") : "";
367
-
368
- text += `<a href="#footnote-ref-${encodeURIComponent(
369
- label,
370
- )}" data-footnote-backref aria-label="Back to reference ${label}"> ↩</a>`;
371
-
372
- this.lexer.state.top = false;
373
- const tokens = this.lexer.blockTokens(text, []);
374
-
375
- const token: Tokens["Footnote"] = {
376
- type: "footnote",
377
- raw: cap[0],
378
- text: text,
379
- label: label,
380
- content: tokens,
381
- sourceMap: this.lexer.getSourceMap(cap[0]),
382
- };
383
-
384
- return token;
385
- }
386
-
387
- html(src: string): Tokens["HTML"] | undefined {
388
- const cap = block.html.exec(src);
389
- if (!cap) return undefined;
390
-
391
- const token: Tokens["HTML"] = {
392
- type: "html",
393
- block: true,
394
- raw: cap[0],
395
- pre: cap[1] === "pre" || cap[1] === "script" || cap[1] === "style",
396
- text: cap[0],
397
- sourceMap: this.lexer.getSourceMap(cap[0]),
398
- };
399
-
400
- /*
401
- * Sometimes an html token does not contain its closing tag.
402
- *
403
- * The following markdown:
404
- *
405
- * 130 <details>
406
- * 131 <summary><h4>Hello World</h4></summary>
407
- * 132
408
- * 133 Here some non-html markdown
409
- * 134 </details>
410
- *
411
- * would result in 3 tokens:
412
- * [
413
- * {
414
- * type: "html",
415
- * raw: "<details>↵ <summary><h4>PUT</h4></summary>↵↵",
416
- * sourceMap: [130, 131]
417
- * },
418
- * {
419
- * type: "paragraph",
420
- * raw: "Here some non-html markdown",
421
- * sourceMap: [133, 133]
422
- * },
423
- * {
424
- * type: "html",
425
- * raw: "</details>",
426
- * }
427
- * ]
428
- *
429
- * sourceMap metadata in first token will be incorrect, because it will
430
- * not take its children into consideration when it should.
431
- *
432
- * To solve that we keep track of html tokens that are pending to be closed
433
- * and update their sourceMap once they're closed.
434
- */
435
-
436
- const capEndsWith = (str?: string) => str && cap[0].trimEnd().endsWith(str);
437
-
438
- const tag = inline.tag.exec(src);
439
- const isHtmlClosed = capEndsWith(tag?.[0].slice(1));
440
-
441
- if (tag?.[0] && !isHtmlClosed) {
442
- // index where the token we just created will be inserted
443
- const tokenIdx = this.lexer.tokens.length;
444
- // first in last out
445
- this.pendingHtmlClose.unshift([tag[0], tokenIdx]);
446
- } else if (this.pendingHtmlClose.length) {
447
- for (const [pendingTag, index] of this.pendingHtmlClose) {
448
- if (capEndsWith(pendingTag.slice(1))) {
449
- const updateToken = this.lexer.tokens[index] as Tokens["HTML"];
450
- if (updateToken.sourceMap?.[1] && token.sourceMap?.[1]) {
451
- updateToken.sourceMap[1] = token.sourceMap[1];
452
- }
453
- this.pendingHtmlClose.shift();
454
- }
447
+ }
448
+ }
449
+
450
+ return token;
451
+ }
452
+
453
+ def(src: string): Tokens["Def"] | undefined {
454
+ const cap = block.def.exec(src);
455
+ if (!cap) return undefined;
456
+
457
+ const tag = cap[1]!.toLowerCase().replace(/\s+/g, " ");
458
+ const href = cap[2]
459
+ ? cap[2].replace(/^<(.*)>$/, "$1").replace(inline.anyPunctuation, "$1")
460
+ : "";
461
+ const title = cap[3]
462
+ ? cap[3].substring(1, cap[3].length - 1).replace(inline.anyPunctuation, "$1")
463
+ : "";
464
+ return {
465
+ type: "def",
466
+ tag,
467
+ raw: cap[0],
468
+ href,
469
+ title,
470
+ sourceMap: this.lexer.getSourceMap(cap[0]),
471
+ };
472
+ }
473
+
474
+ table(src: string): Tokens["Table"] | undefined {
475
+ const cap = block.table.exec(src);
476
+ if (!cap?.[2]) return;
477
+
478
+ if (!/[:|]/.test(cap[2])) {
479
+ // delimiter row must have a pipe (|) or colon (:) otherwise it is a setext heading
480
+ return;
481
+ }
482
+
483
+ const item: Tokens["Table"] = {
484
+ type: "table",
485
+ raw: cap[0],
486
+ header: splitCells(cap[1]!).map((c) => ({
487
+ type: "tablecell",
488
+ raw: c,
489
+ text: c,
490
+ tokens: [],
491
+ })),
492
+ align: [],
493
+ rows: [],
494
+ sourceMap: this.lexer.getSourceMap(cap[0]),
495
+ };
496
+
497
+ const align = cap[2].replace(/^\||\| *$/g, "").split("|") as (string | null)[];
498
+ const rows = cap[3]?.trim() ? cap[3].replace(/\n[ \t]*$/, "").split("\n") : [];
499
+
500
+ if (item.header.length !== align.length) return;
501
+
502
+ let l = align.length;
503
+ let i, j, k, row;
504
+ for (i = 0; i < l; i++) {
505
+ const alignStr = align[i];
506
+ if (alignStr) {
507
+ if (/^ *-+: *$/.test(alignStr)) {
508
+ item.align.push("right");
509
+ } else if (/^ *:-+: *$/.test(alignStr)) {
510
+ item.align.push("center");
511
+ } else if (/^ *:-+ *$/.test(alignStr)) {
512
+ item.align.push("left");
513
+ } else {
514
+ item.align.push(null);
455
515
  }
456
- }
457
-
458
- return token;
459
- }
460
-
461
- def(src: string): Tokens["Def"] | undefined {
462
- const cap = block.def.exec(src);
463
- if (!cap) return undefined;
464
-
465
- const tag = cap[1]!.toLowerCase().replace(/\s+/g, " ");
466
- const href = cap[2]
467
- ? cap[2].replace(/^<(.*)>$/, "$1").replace(inline.anyPunctuation, "$1")
468
- : "";
469
- const title = cap[3]
470
- ? cap[3].substring(1, cap[3].length - 1).replace(inline.anyPunctuation, "$1")
471
- : "";
472
- return {
473
- type: "def",
474
- tag,
475
- raw: cap[0],
476
- href,
477
- title,
478
- sourceMap: this.lexer.getSourceMap(cap[0]),
479
- };
480
- }
481
-
482
- table(src: string): Tokens["Table"] | undefined {
483
- const cap = block.table.exec(src);
484
- if (!cap?.[2]) return;
516
+ }
517
+ }
518
+
519
+ l = rows.length;
520
+ for (i = 0; i < l; i++) {
521
+ item.rows.push(
522
+ splitCells(rows[i] as unknown as string, item.header.length).map((c) => ({
523
+ type: "tablecell",
524
+ raw: c,
525
+ text: c,
526
+ tokens: [],
527
+ })),
528
+ );
529
+ }
530
+
531
+ // parse child tokens inside headers and cells
532
+
533
+ // header child tokens
534
+ l = item.header.length;
535
+ for (j = 0; j < l; j++) {
536
+ item.header[j]!.tokens = this.lexer.inline(item.header[j]!.text);
537
+ }
538
+
539
+ // cell child tokens
540
+ l = item.rows.length;
541
+ for (j = 0; j < l; j++) {
542
+ row = item.rows[j]!;
543
+ for (k = 0; k < row.length; k++) {
544
+ row[k]!.tokens = this.lexer.inline(row[k]!.text);
545
+ }
546
+ }
547
+
548
+ return item;
549
+ }
550
+
551
+ lheading(src: string): Tokens["Heading"] | undefined {
552
+ const cap = block.lheading.exec(src);
553
+ if (!cap) return undefined;
554
+
555
+ return {
556
+ type: "heading",
557
+ raw: cap[0],
558
+ depth: cap[2]!.startsWith("=") ? 1 : 2,
559
+ text: cap[1]!,
560
+ tokens: this.lexer.inline(cap[1]!),
561
+ sourceMap: this.lexer.getSourceMap(cap[0]),
562
+ };
563
+ }
564
+
565
+ paragraph(src: string): Tokens["Paragraph"] | undefined {
566
+ const cap = block.paragraph.exec(src);
567
+ if (!cap) return undefined;
568
+
569
+ const text = cap[1]!.endsWith("\n") ? cap[1]!.slice(0, -1) : cap[1]!;
570
+ return {
571
+ type: "paragraph",
572
+ raw: cap[0],
573
+ text,
574
+ tokens: this.lexer.inline(text),
575
+ sourceMap: this.lexer.getSourceMap(cap[0]),
576
+ };
577
+ }
578
+
579
+ text(src: string): Tokens["Text"] | undefined {
580
+ const cap = block.text.exec(src);
581
+ if (!cap) return undefined;
582
+
583
+ return {
584
+ type: "text",
585
+ raw: cap[0],
586
+ text: cap[0],
587
+ tokens: this.lexer.inline(cap[0]),
588
+ sourceMap: this.lexer.getSourceMap(cap[0]),
589
+ };
590
+ }
591
+
592
+ escape(src: string): Tokens["Escape"] | undefined {
593
+ const cap = inline.escape.exec(src);
594
+ if (!cap) return undefined;
595
+
596
+ return {
597
+ type: "escape",
598
+ raw: cap[0],
599
+ text: escape(cap[1]!),
600
+ };
601
+ }
602
+
603
+ tag(src: string): Tokens["Tag"] | undefined {
604
+ const cap = inline.tag.exec(src);
605
+ if (!cap) return undefined;
606
+
607
+ if (!this.lexer.state.inLink && /^<a /i.test(cap[0])) {
608
+ this.lexer.state.inLink = true;
609
+ } else if (this.lexer.state.inLink && /^<\/a>/i.test(cap[0])) {
610
+ this.lexer.state.inLink = false;
611
+ }
612
+ if (!this.lexer.state.inRawBlock && /^<(pre|code|kbd|script)(\s|>)/i.test(cap[0])) {
613
+ this.lexer.state.inRawBlock = true;
614
+ } else if (this.lexer.state.inRawBlock && /^<\/(pre|code|kbd|script)(\s|>)/i.test(cap[0])) {
615
+ this.lexer.state.inRawBlock = false;
616
+ }
617
+
618
+ return {
619
+ type: "html",
620
+ raw: cap[0],
621
+ inLink: this.lexer.state.inLink,
622
+ inRawBlock: this.lexer.state.inRawBlock,
623
+ block: false,
624
+ text: cap[0],
625
+ };
626
+ }
627
+
628
+ link(src: string): Tokens["Link"] | Tokens["Image"] | undefined {
629
+ const cap = inline.link.exec(src);
630
+ if (!cap) return undefined;
631
+
632
+ const trimmedUrl = cap[2]!.trim();
633
+ if (trimmedUrl.startsWith("<")) {
634
+ // commonmark requires matching angle brackets
635
+ if (!trimmedUrl.endsWith(">")) {
636
+ return;
637
+ }
485
638
 
486
- if (!/[:|]/.test(cap[2])) {
487
- // delimiter row must have a pipe (|) or colon (:) otherwise it is a setext heading
639
+ // ending angle bracket cannot be escaped
640
+ const rtrimSlash = rtrim(trimmedUrl.slice(0, -1), "\\");
641
+ if ((trimmedUrl.length - rtrimSlash.length) % 2 === 0) {
488
642
  return;
489
- }
643
+ }
644
+ } else {
645
+ // find closing parenthesis
646
+ const lastParenIndex = findClosingBracket(cap[2]!, "()");
647
+ if (lastParenIndex > -1) {
648
+ const start = cap[0].startsWith("!") ? 5 : 4;
649
+ const linkLen = start + cap[1]!.length + lastParenIndex;
650
+ cap[2] = cap[2]!.substring(0, lastParenIndex);
651
+ cap[0] = cap[0].substring(0, linkLen).trim();
652
+ cap[3] = "";
653
+ }
654
+ }
655
+ let href = cap[2]!;
656
+ let title = "";
657
+ title = cap[3] ? cap[3].slice(1, -1) : "";
658
+
659
+ href = href.trim();
660
+ if (href.startsWith("<")) {
661
+ href = href.slice(1, -1);
662
+ }
663
+ return outputLink(
664
+ cap,
665
+ {
666
+ href: href ? href.replace(inline.anyPunctuation, "$1") : href,
667
+ title: title ? title.replace(inline.anyPunctuation, "$1") : title,
668
+ },
669
+ cap[0],
670
+ this.lexer,
671
+ );
672
+ }
673
+
674
+ reflink(
675
+ src: string,
676
+ links: Links,
677
+ ): Tokens["Link"] | Tokens["Image"] | Tokens["Text"] | undefined {
678
+ let cap;
679
+ if ((cap = inline.reflink.exec(src)) ?? (cap = inline.nolink.exec(src))) {
680
+ const linkStr = (cap[2] ?? cap[1])!.replace(/\s+/g, " ");
681
+ const link = links[linkStr.toLowerCase()];
682
+ if (!link) {
683
+ const text = cap[0].charAt(0);
684
+ return {
685
+ type: "text",
686
+ raw: text,
687
+ text,
688
+ };
689
+ }
690
+ return outputLink(cap, link, cap[0], this.lexer);
691
+ }
692
+ return undefined;
693
+ }
694
+
695
+ emStrong(
696
+ src: string,
697
+ maskedSrc: string,
698
+ prevChar = "",
699
+ ): Tokens["Em"] | Tokens["Strong"] | undefined {
700
+ let match = inline.emStrong.lDelim.exec(src);
701
+ if (!match) return;
702
+
703
+ // _ can't be between two alphanumerics. \p{L}\p{N} includes non-english alphabet/numbers as well
704
+ if (match[3] && /[\p{L}\p{N}]/u.exec(prevChar)) return;
705
+
706
+ // eslint-disable-next-line
707
+ const nextChar = match[1] || match[2] || "";
708
+
709
+ if (!nextChar || !prevChar || inline.punctuation.exec(prevChar)) {
710
+ // unicode Regex counts emoji as 1 char; spread into array for proper count (used multiple times below)
711
+ // eslint-disable-next-line @typescript-eslint/no-misused-spread
712
+ const lLength = [...match[0]].length - 1;
713
+ let rDelim,
714
+ rLength,
715
+ delimTotal = lLength,
716
+ midDelimTotal = 0;
717
+
718
+ const endReg = match[0].startsWith("*")
719
+ ? inline.emStrong.rDelimAst
720
+ : inline.emStrong.rDelimUnd;
721
+ endReg.lastIndex = 0;
722
+
723
+ // Clip maskedSrc to same section of string as src (move to lexer?)
724
+ maskedSrc = maskedSrc.slice(-1 * src.length + lLength);
725
+
726
+ while ((match = endReg.exec(maskedSrc)) != null) {
727
+ // eslint-disable-next-line
728
+ rDelim = match[1] || match[2] || match[3] || match[4] || match[5] || match[6];
729
+
730
+ if (!rDelim) continue; // skip single * in __abc*abc__
490
731
 
491
- const item: Tokens["Table"] = {
492
- type: "table",
493
- raw: cap[0],
494
- header: splitCells(cap[1]!).map((c) => ({
495
- type: "tablecell",
496
- raw: c,
497
- text: c,
498
- tokens: [],
499
- })),
500
- align: [],
501
- rows: [],
502
- sourceMap: this.lexer.getSourceMap(cap[0]),
503
- };
504
-
505
- const align = cap[2].replace(/^\||\| *$/g, "").split("|") as (string | null)[];
506
- const rows = cap[3]?.trim() ? cap[3].replace(/\n[ \t]*$/, "").split("\n") : [];
507
-
508
- if (item.header.length !== align.length) return;
509
-
510
- let l = align.length;
511
- let i, j, k, row;
512
- for (i = 0; i < l; i++) {
513
- const alignStr = align[i];
514
- if (alignStr) {
515
- if (/^ *-+: *$/.test(alignStr)) {
516
- item.align.push("right");
517
- } else if (/^ *:-+: *$/.test(alignStr)) {
518
- item.align.push("center");
519
- } else if (/^ *:-+ *$/.test(alignStr)) {
520
- item.align.push("left");
521
- } else {
522
- item.align.push(null);
523
- }
524
- }
525
- }
526
-
527
- l = rows.length;
528
- for (i = 0; i < l; i++) {
529
- item.rows.push(
530
- splitCells(rows[i] as unknown as string, item.header.length).map((c) => ({
531
- type: "tablecell",
532
- raw: c,
533
- text: c,
534
- tokens: [],
535
- })),
536
- );
537
- }
538
-
539
- // parse child tokens inside headers and cells
540
-
541
- // header child tokens
542
- l = item.header.length;
543
- for (j = 0; j < l; j++) {
544
- item.header[j]!.tokens = this.lexer.inline(item.header[j]!.text);
545
- }
546
-
547
- // cell child tokens
548
- l = item.rows.length;
549
- for (j = 0; j < l; j++) {
550
- row = item.rows[j]!;
551
- for (k = 0; k < row.length; k++) {
552
- row[k]!.tokens = this.lexer.inline(row[k]!.text);
732
+ // eslint-disable-next-line @typescript-eslint/no-misused-spread
733
+ rLength = [...rDelim].length;
734
+
735
+ if (match[3] || match[4]) {
736
+ // found another Left Delim
737
+ delimTotal += rLength;
738
+ continue;
739
+ } else if (match[5] || match[6]) {
740
+ // either Left or Right Delim
741
+ if (lLength % 3 && !((lLength + rLength) % 3)) {
742
+ midDelimTotal += rLength;
743
+ continue; // CommonMark Emphasis Rules 9-10
744
+ }
553
745
  }
554
- }
555
746
 
556
- return item;
557
- }
747
+ delimTotal -= rLength;
558
748
 
559
- lheading(src: string): Tokens["Heading"] | undefined {
560
- const cap = block.lheading.exec(src);
561
- if (!cap) return undefined;
749
+ if (delimTotal > 0) continue; // Haven't found enough closing delimiters
562
750
 
563
- return {
564
- type: "heading",
565
- raw: cap[0],
566
- depth: cap[2]!.startsWith("=") ? 1 : 2,
567
- text: cap[1]!,
568
- tokens: this.lexer.inline(cap[1]!),
569
- sourceMap: this.lexer.getSourceMap(cap[0]),
570
- };
571
- }
572
-
573
- paragraph(src: string): Tokens["Paragraph"] | undefined {
574
- const cap = block.paragraph.exec(src);
575
- if (!cap) return undefined;
576
-
577
- const text = cap[1]!.endsWith("\n") ? cap[1]!.slice(0, -1) : cap[1]!;
578
- return {
579
- type: "paragraph",
580
- raw: cap[0],
581
- text,
582
- tokens: this.lexer.inline(text),
583
- sourceMap: this.lexer.getSourceMap(cap[0]),
584
- };
585
- }
586
-
587
- text(src: string): Tokens["Text"] | undefined {
588
- const cap = block.text.exec(src);
589
- if (!cap) return undefined;
590
-
591
- return {
592
- type: "text",
593
- raw: cap[0],
594
- text: cap[0],
595
- tokens: this.lexer.inline(cap[0]),
596
- sourceMap: this.lexer.getSourceMap(cap[0]),
597
- };
598
- }
599
-
600
- escape(src: string): Tokens["Escape"] | undefined {
601
- const cap = inline.escape.exec(src);
602
- if (!cap) return undefined;
603
-
604
- return {
605
- type: "escape",
606
- raw: cap[0],
607
- text: escape(cap[1]!),
608
- };
609
- }
610
-
611
- tag(src: string): Tokens["Tag"] | undefined {
612
- const cap = inline.tag.exec(src);
613
- if (!cap) return undefined;
614
-
615
- if (!this.lexer.state.inLink && /^<a /i.test(cap[0])) {
616
- this.lexer.state.inLink = true;
617
- } else if (this.lexer.state.inLink && /^<\/a>/i.test(cap[0])) {
618
- this.lexer.state.inLink = false;
619
- }
620
- if (!this.lexer.state.inRawBlock && /^<(pre|code|kbd|script)(\s|>)/i.test(cap[0])) {
621
- this.lexer.state.inRawBlock = true;
622
- } else if (this.lexer.state.inRawBlock && /^<\/(pre|code|kbd|script)(\s|>)/i.test(cap[0])) {
623
- this.lexer.state.inRawBlock = false;
624
- }
625
-
626
- return {
627
- type: "html",
628
- raw: cap[0],
629
- inLink: this.lexer.state.inLink,
630
- inRawBlock: this.lexer.state.inRawBlock,
631
- block: false,
632
- text: cap[0],
633
- };
634
- }
635
-
636
- link(src: string): Tokens["Link"] | Tokens["Image"] | undefined {
637
- const cap = inline.link.exec(src);
638
- if (!cap) return undefined;
639
-
640
- const trimmedUrl = cap[2]!.trim();
641
- if (trimmedUrl.startsWith("<")) {
642
- // commonmark requires matching angle brackets
643
- if (!trimmedUrl.endsWith(">")) {
644
- return;
751
+ // Remove extra characters. *a*** -> *a*
752
+ rLength = Math.min(rLength, rLength + delimTotal + midDelimTotal);
753
+ // char length can be >1 for unicode characters;
754
+ // eslint-disable-next-line @typescript-eslint/no-misused-spread
755
+ const lastCharLength = [...match[0]][0]!.length;
756
+ const raw = src.slice(0, lLength + match.index + lastCharLength + rLength);
757
+
758
+ // Create `em` if smallest delimiter has odd char count. *a***
759
+ if (Math.min(lLength, rLength) % 2) {
760
+ const text = raw.slice(1, -1);
761
+ return {
762
+ type: "em",
763
+ raw,
764
+ text,
765
+ tokens: this.lexer.inlineTokens(text),
766
+ };
645
767
  }
646
768
 
647
- // ending angle bracket cannot be escaped
648
- const rtrimSlash = rtrim(trimmedUrl.slice(0, -1), "\\");
649
- if ((trimmedUrl.length - rtrimSlash.length) % 2 === 0) {
650
- return;
651
- }
652
- } else {
653
- // find closing parenthesis
654
- const lastParenIndex = findClosingBracket(cap[2]!, "()");
655
- if (lastParenIndex > -1) {
656
- const start = cap[0].startsWith("!") ? 5 : 4;
657
- const linkLen = start + cap[1]!.length + lastParenIndex;
658
- cap[2] = cap[2]!.substring(0, lastParenIndex);
659
- cap[0] = cap[0].substring(0, linkLen).trim();
660
- cap[3] = "";
661
- }
662
- }
663
- let href = cap[2]!;
664
- let title = "";
665
- title = cap[3] ? cap[3].slice(1, -1) : "";
666
-
667
- href = href.trim();
668
- if (href.startsWith("<")) {
669
- href = href.slice(1, -1);
670
- }
671
- return outputLink(
672
- cap,
769
+ // Create 'strong' if smallest delimiter has even char count. **a***
770
+ const text = raw.slice(2, -2);
771
+ return {
772
+ type: "strong",
773
+ raw,
774
+ text,
775
+ tokens: this.lexer.inlineTokens(text),
776
+ };
777
+ }
778
+ }
779
+
780
+ return undefined;
781
+ }
782
+
783
+ footnoteRef(src: string): Tokens["FootnoteRef"] | undefined {
784
+ const cap = inline.footnoteRef.exec(src);
785
+ if (!cap) return undefined;
786
+
787
+ return {
788
+ type: "footnoteRef",
789
+ raw: cap[0],
790
+ label: cap[1] ?? "",
791
+ };
792
+ }
793
+
794
+ codespan(src: string): Tokens["Codespan"] | undefined {
795
+ const cap = inline.code.exec(src);
796
+ if (!cap) return undefined;
797
+
798
+ let text = cap[2]!.replace(/\n/g, " ");
799
+ const hasNonSpaceChars = /[^ ]/.test(text);
800
+ const hasSpaceCharsOnBothEnds = text.startsWith(" ") && text.endsWith(" ");
801
+ if (hasNonSpaceChars && hasSpaceCharsOnBothEnds) {
802
+ text = text.substring(1, text.length - 1);
803
+ }
804
+ text = escape(text, true);
805
+ return {
806
+ type: "codespan",
807
+ raw: cap[0],
808
+ text,
809
+ };
810
+ }
811
+
812
+ br(src: string): Tokens["Br"] | undefined {
813
+ const cap = inline.br.exec(src);
814
+ if (!cap) return undefined;
815
+
816
+ return {
817
+ type: "br",
818
+ raw: cap[0],
819
+ };
820
+ }
821
+
822
+ del(src: string): Tokens["Del"] | undefined {
823
+ const cap = inline.del.exec(src);
824
+ if (!cap) return undefined;
825
+
826
+ return {
827
+ type: "del",
828
+ raw: cap[0],
829
+ text: cap[2]!,
830
+ tokens: this.lexer.inlineTokens(cap[2]!),
831
+ };
832
+ }
833
+
834
+ autolink(src: string): Tokens["Link"] | undefined {
835
+ const cap = inline.autolink.exec(src);
836
+ if (!cap) return undefined;
837
+
838
+ let text, href;
839
+ if (cap[2] === "@") {
840
+ text = escape(cap[1]!);
841
+ href = "mailto:" + text;
842
+ } else {
843
+ text = escape(cap[1]!);
844
+ href = text;
845
+ }
846
+
847
+ return {
848
+ type: "link",
849
+ title: null,
850
+ raw: cap[0],
851
+ text,
852
+ href,
853
+ tokens: [
673
854
  {
674
- href: href ? href.replace(inline.anyPunctuation, "$1") : href,
675
- title: title ? title.replace(inline.anyPunctuation, "$1") : title,
855
+ type: "text",
856
+ raw: text,
857
+ text,
676
858
  },
677
- cap[0],
678
- this.lexer,
679
- );
680
- }
681
-
682
- reflink(
683
- src: string,
684
- links: Links,
685
- ): Tokens["Link"] | Tokens["Image"] | Tokens["Text"] | undefined {
686
- let cap;
687
- if ((cap = inline.reflink.exec(src)) ?? (cap = inline.nolink.exec(src))) {
688
- const linkStr = (cap[2] ?? cap[1])!.replace(/\s+/g, " ");
689
- const link = links[linkStr.toLowerCase()];
690
- if (!link) {
691
- const text = cap[0].charAt(0);
692
- return {
693
- type: "text",
694
- raw: text,
695
- text,
696
- };
697
- }
698
- return outputLink(cap, link, cap[0], this.lexer);
699
- }
700
- return undefined;
701
- }
702
-
703
- emStrong(
704
- src: string,
705
- maskedSrc: string,
706
- prevChar = "",
707
- ): Tokens["Em"] | Tokens["Strong"] | undefined {
708
- let match = inline.emStrong.lDelim.exec(src);
709
- if (!match) return;
710
-
711
- // _ can't be between two alphanumerics. \p{L}\p{N} includes non-english alphabet/numbers as well
712
- if (match[3] && /[\p{L}\p{N}]/u.exec(prevChar)) return;
713
-
714
- // eslint-disable-next-line
715
- const nextChar = match[1] || match[2] || "";
716
-
717
- if (!nextChar || !prevChar || inline.punctuation.exec(prevChar)) {
718
- // unicode Regex counts emoji as 1 char; spread into array for proper count (used multiple times below)
719
- // eslint-disable-next-line @typescript-eslint/no-misused-spread
720
- const lLength = [...match[0]].length - 1;
721
- let rDelim,
722
- rLength,
723
- delimTotal = lLength,
724
- midDelimTotal = 0;
725
-
726
- const endReg = match[0].startsWith("*")
727
- ? inline.emStrong.rDelimAst
728
- : inline.emStrong.rDelimUnd;
729
- endReg.lastIndex = 0;
730
-
731
- // Clip maskedSrc to same section of string as src (move to lexer?)
732
- maskedSrc = maskedSrc.slice(-1 * src.length + lLength);
733
-
734
- while ((match = endReg.exec(maskedSrc)) != null) {
735
- // eslint-disable-next-line
736
- rDelim = match[1] || match[2] || match[3] || match[4] || match[5] || match[6];
737
-
738
- if (!rDelim) continue; // skip single * in __abc*abc__
739
-
740
- // eslint-disable-next-line @typescript-eslint/no-misused-spread
741
- rLength = [...rDelim].length;
742
-
743
- if (match[3] || match[4]) {
744
- // found another Left Delim
745
- delimTotal += rLength;
746
- continue;
747
- } else if (match[5] || match[6]) {
748
- // either Left or Right Delim
749
- if (lLength % 3 && !((lLength + rLength) % 3)) {
750
- midDelimTotal += rLength;
751
- continue; // CommonMark Emphasis Rules 9-10
752
- }
753
- }
754
-
755
- delimTotal -= rLength;
756
-
757
- if (delimTotal > 0) continue; // Haven't found enough closing delimiters
758
-
759
- // Remove extra characters. *a*** -> *a*
760
- rLength = Math.min(rLength, rLength + delimTotal + midDelimTotal);
761
- // char length can be >1 for unicode characters;
762
- // eslint-disable-next-line @typescript-eslint/no-misused-spread
763
- const lastCharLength = [...match[0]][0]!.length;
764
- const raw = src.slice(0, lLength + match.index + lastCharLength + rLength);
765
-
766
- // Create `em` if smallest delimiter has odd char count. *a***
767
- if (Math.min(lLength, rLength) % 2) {
768
- const text = raw.slice(1, -1);
769
- return {
770
- type: "em",
771
- raw,
772
- text,
773
- tokens: this.lexer.inlineTokens(text),
774
- };
775
- }
776
-
777
- // Create 'strong' if smallest delimiter has even char count. **a***
778
- const text = raw.slice(2, -2);
779
- return {
780
- type: "strong",
781
- raw,
782
- text,
783
- tokens: this.lexer.inlineTokens(text),
784
- };
785
- }
786
- }
787
-
788
- return undefined;
789
- }
790
-
791
- footnoteRef(src: string): Tokens["FootnoteRef"] | undefined {
792
- const cap = inline.footnoteRef.exec(src);
793
- if (!cap) return undefined;
794
-
795
- return {
796
- type: "footnoteRef",
797
- raw: cap[0],
798
- label: cap[1] ?? "",
799
- };
800
- }
801
-
802
- codespan(src: string): Tokens["Codespan"] | undefined {
803
- const cap = inline.code.exec(src);
804
- if (!cap) return undefined;
805
-
806
- let text = cap[2]!.replace(/\n/g, " ");
807
- const hasNonSpaceChars = /[^ ]/.test(text);
808
- const hasSpaceCharsOnBothEnds = text.startsWith(" ") && text.endsWith(" ");
809
- if (hasNonSpaceChars && hasSpaceCharsOnBothEnds) {
810
- text = text.substring(1, text.length - 1);
811
- }
812
- text = escape(text, true);
813
- return {
814
- type: "codespan",
815
- raw: cap[0],
816
- text,
817
- };
818
- }
819
-
820
- br(src: string): Tokens["Br"] | undefined {
821
- const cap = inline.br.exec(src);
822
- if (!cap) return undefined;
823
-
824
- return {
825
- type: "br",
826
- raw: cap[0],
827
- };
828
- }
829
-
830
- del(src: string): Tokens["Del"] | undefined {
831
- const cap = inline.del.exec(src);
832
- if (!cap) return undefined;
833
-
834
- return {
835
- type: "del",
836
- raw: cap[0],
837
- text: cap[2]!,
838
- tokens: this.lexer.inlineTokens(cap[2]!),
839
- };
840
- }
841
-
842
- autolink(src: string): Tokens["Link"] | undefined {
843
- const cap = inline.autolink.exec(src);
844
- if (!cap) return undefined;
845
-
846
- let text, href;
847
- if (cap[2] === "@") {
848
- text = escape(cap[1]!);
859
+ ],
860
+ };
861
+ }
862
+
863
+ url(src: string): Tokens["Link"] | undefined {
864
+ let cap;
865
+ if ((cap = inline.url.exec(src))) {
866
+ let text, href;
867
+ if (cap[2] === "@") {
868
+ text = escape(cap[0]);
849
869
  href = "mailto:" + text;
850
- } else {
851
- text = escape(cap[1]!);
852
- href = text;
853
- }
854
-
855
- return {
870
+ } else {
871
+ // do extended autolink path validation
872
+ let prevCapZero;
873
+ do {
874
+ prevCapZero = cap[0];
875
+ cap[0] = inline.backpedal.exec(cap[0])![0];
876
+ } while (prevCapZero !== cap[0]);
877
+ text = escape(cap[0]);
878
+ if (cap[1] === "www.") {
879
+ href = "http://" + cap[0];
880
+ } else {
881
+ href = cap[0];
882
+ }
883
+ }
884
+ return {
856
885
  type: "link",
857
886
  title: null,
858
887
  raw: cap[0],
859
888
  text,
860
889
  href,
861
890
  tokens: [
862
- {
863
- type: "text",
864
- raw: text,
865
- text,
866
- },
891
+ {
892
+ type: "text",
893
+ raw: text,
894
+ text,
895
+ },
867
896
  ],
868
- };
869
- }
870
-
871
- url(src: string): Tokens["Link"] | undefined {
872
- let cap;
873
- if ((cap = inline.url.exec(src))) {
874
- let text, href;
875
- if (cap[2] === "@") {
876
- text = escape(cap[0]);
877
- href = "mailto:" + text;
878
- } else {
879
- // do extended autolink path validation
880
- let prevCapZero;
881
- do {
882
- prevCapZero = cap[0];
883
- cap[0] = inline.backpedal.exec(cap[0])![0];
884
- } while (prevCapZero !== cap[0]);
885
- text = escape(cap[0]);
886
- if (cap[1] === "www.") {
887
- href = "http://" + cap[0];
888
- } else {
889
- href = cap[0];
890
- }
891
- }
892
- return {
893
- type: "link",
894
- title: null,
895
- raw: cap[0],
896
- text,
897
- href,
898
- tokens: [
899
- {
900
- type: "text",
901
- raw: text,
902
- text,
903
- },
904
- ],
905
- };
906
- }
907
- return undefined;
908
- }
909
-
910
- inlineText(src: string): Tokens["Text"] | undefined {
911
- const cap = inline.text.exec(src);
912
- if (!cap) return undefined;
913
-
914
- let text;
915
- if (this.lexer.state.inRawBlock) {
916
- text = cap[0];
917
- } else {
918
- text = escape(cap[0]);
919
- }
920
- return {
921
- type: "text",
922
- raw: cap[0],
923
- text,
924
- };
925
- }
926
-
927
- latexBlock(src: string): Tokens["LatexBlock"] | undefined {
928
- const cap = block.latexBlock.exec(src);
929
- if (!cap) return undefined;
930
-
931
- // cap[1] is from $$...$$ syntax, cap[2] is from \[...\] syntax
932
- const text = cap[1] ?? cap[2] ?? "";
933
-
934
- return {
935
- type: "latexBlock",
936
- raw: cap[0],
937
- text: text.trim(),
938
- sourceMap: this.lexer.getSourceMap(cap[0]),
939
- };
940
- }
941
-
942
- latexInline(src: string): Tokens["LatexInline"] | undefined {
943
- const cap = inline.latexInline.exec(src);
944
- if (!cap) return undefined;
945
-
946
- // cap[1] is from $...$ syntax, cap[2] is from \(...\) syntax
947
- const text = cap[1] ?? cap[2] ?? "";
948
-
949
- return {
950
- type: "latexInline",
951
- raw: cap[0],
952
- text,
953
- };
954
- }
897
+ };
898
+ }
899
+ return undefined;
900
+ }
901
+
902
+ inlineText(src: string): Tokens["Text"] | undefined {
903
+ const cap = inline.text.exec(src);
904
+ if (!cap) return undefined;
905
+
906
+ let text;
907
+ if (this.lexer.state.inRawBlock) {
908
+ text = cap[0];
909
+ } else {
910
+ text = escape(cap[0]);
911
+ }
912
+ return {
913
+ type: "text",
914
+ raw: cap[0],
915
+ text,
916
+ };
917
+ }
918
+
919
+ latexBlock(src: string): Tokens["LatexBlock"] | undefined {
920
+ const cap = block.latexBlock.exec(src);
921
+ if (!cap) return undefined;
922
+
923
+ // cap[1] is from $$...$$ syntax, cap[2] is from \[...\] syntax
924
+ const text = cap[1] ?? cap[2] ?? "";
925
+
926
+ return {
927
+ type: "latexBlock",
928
+ raw: cap[0],
929
+ text: text.trim(),
930
+ sourceMap: this.lexer.getSourceMap(cap[0]),
931
+ };
932
+ }
933
+
934
+ latexInline(src: string): Tokens["LatexInline"] | undefined {
935
+ const cap = inline.latexInline.exec(src);
936
+ if (!cap) return undefined;
937
+
938
+ // cap[1] is from $...$ syntax, cap[2] is from \(...\) syntax
939
+ const text = cap[1] ?? cap[2] ?? "";
940
+
941
+ return {
942
+ type: "latexInline",
943
+ raw: cap[0],
944
+ text,
945
+ };
946
+ }
955
947
  }