pantsdown 2.2.6 → 2.2.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/tokenizer.ts CHANGED
@@ -1,15 +1,17 @@
1
1
  import type { Lexer } from "./lexer.ts";
2
2
  import { block } from "./rules/block.ts";
3
3
  import { inline } from "./rules/inline.ts";
4
- import { type Links, type Tokens } from "./types.ts";
4
+ import { other } from "./rules/other.ts";
5
+ import { type Links, type Token, type Tokens } from "./types.ts";
5
6
  import {
6
7
  ALERTS,
7
- escape,
8
+ expandTabs,
8
9
  findClosingBracket,
9
10
  indentCodeCompensation,
10
11
  outputLink,
11
12
  rtrim,
12
13
  splitCells,
14
+ trimTrailingBlankLines,
13
15
  } from "./utils.ts";
14
16
 
15
17
  /**
@@ -40,13 +42,14 @@ export class Tokenizer {
40
42
  const cap = block.code.exec(src);
41
43
  if (!cap) return undefined;
42
44
 
43
- const text = cap[0].replace(/^ {1,4}/gm, "");
45
+ const raw = trimTrailingBlankLines(cap[0]);
46
+ const text = raw.replace(other.codeRemoveIndent, "");
44
47
  return {
45
48
  type: "code",
46
- raw: cap[0],
49
+ raw,
47
50
  codeBlockStyle: "indented",
48
- text: rtrim(text, "\n"),
49
- sourceMap: this.lexer.getSourceMap(cap[0]),
51
+ text,
52
+ sourceMap: this.lexer.getSourceMap(raw),
50
53
  };
51
54
  }
52
55
 
@@ -80,13 +83,14 @@ export class Tokenizer {
80
83
  }
81
84
  }
82
85
 
86
+ const raw = rtrim(cap[0], "\n");
83
87
  return {
84
88
  type: "heading",
85
- raw: cap[0],
89
+ raw,
86
90
  depth: cap[1]!.length,
87
91
  text,
88
92
  tokens: this.lexer.inline(text),
89
- sourceMap: this.lexer.getSourceMap(cap[0]),
93
+ sourceMap: this.lexer.getSourceMap(raw),
90
94
  };
91
95
  }
92
96
 
@@ -94,10 +98,11 @@ export class Tokenizer {
94
98
  const cap = block.hr.exec(src);
95
99
  if (!cap) return undefined;
96
100
 
101
+ const raw = rtrim(cap[0], "\n");
97
102
  return {
98
103
  type: "hr",
99
- raw: cap[0],
100
- sourceMap: this.lexer.getSourceMap(cap[0]),
104
+ raw,
105
+ sourceMap: this.lexer.getSourceMap(raw),
101
106
  };
102
107
  }
103
108
 
@@ -105,18 +110,95 @@ export class Tokenizer {
105
110
  const cap = block.blockquote.exec(src);
106
111
  if (!cap) return undefined;
107
112
 
108
- // precede setext continuation with 4 spaces so it isn't a setext
109
- let text = cap[0].replace(/\n {0,3}((?:=+|-+) *)(?=\n|$)/g, "\n $1");
110
- text = rtrim(text.replace(/^ *>[ \t]?/gm, ""), "\n");
111
- const top = this.lexer.state.top;
112
- this.lexer.state.top = true;
113
- const tokens = this.lexer.blockTokens(text, []);
114
- this.lexer.state.top = top;
115
- this.lexer.line++;
113
+ let lines = rtrim(cap[0], "\n").split("\n");
114
+ let raw = "";
115
+ let text = "";
116
+ const tokens: Token[] = [];
117
+ // the accumulated `raw` can end up with garbled *content* at continuation-merge
118
+ // boundaries (upstream does the same), but its length and newline count always
119
+ // match the consumed source — which is all the lexer and sourcemaps rely on
120
+ const startLine = this.lexer.line;
121
+
122
+ while (lines.length > 0) {
123
+ let inBlockquote = false;
124
+ const currentLines = [];
125
+
126
+ let i;
127
+ for (i = 0; i < lines.length; i++) {
128
+ // get lines up to a continuation
129
+ if (other.blockquoteStart.test(lines[i]!)) {
130
+ currentLines.push(lines[i]!);
131
+ inBlockquote = true;
132
+ } else if (!inBlockquote) {
133
+ currentLines.push(lines[i]!);
134
+ } else {
135
+ break;
136
+ }
137
+ }
138
+ lines = lines.slice(i);
139
+
140
+ const currentRaw = currentLines.join("\n");
141
+ const currentText = currentRaw
142
+ // precede setext continuation with 4 spaces so it isn't a setext
143
+ .replace(other.blockquoteSetextReplace, "\n $1")
144
+ .replace(other.blockquoteSetextReplace2, "");
145
+ // this round starts right after the lines already accumulated in `raw`
146
+ this.lexer.line = startLine + (raw ? raw.split("\n").length : 0);
147
+ raw = raw ? `${raw}\n${currentRaw}` : currentRaw;
148
+ text = text ? `${text}\n${currentText}` : currentText;
149
+
150
+ // parse blockquote lines as top level tokens
151
+ // merge paragraphs if this is a continuation
152
+ const top = this.lexer.state.top;
153
+ this.lexer.state.top = true;
154
+ this.lexer.blockTokens(currentText, tokens, true);
155
+ this.lexer.state.top = top;
156
+
157
+ // if there is no continuation then we are done
158
+ if (lines.length === 0) {
159
+ break;
160
+ }
161
+
162
+ const lastToken = tokens[tokens.length - 1];
163
+
164
+ if (lastToken?.type === "code") {
165
+ // blockquote continuation cannot be preceded by a code block
166
+ break;
167
+ } else if (lastToken?.type === "blockquote") {
168
+ // include continuation in nested blockquote
169
+ const oldToken = lastToken;
170
+ const newText = oldToken.raw + "\n" + lines.join("\n");
171
+ // re-lexing the nested blockquote restarts at its first source line
172
+ this.lexer.line = startLine + raw.split("\n").length - oldToken.raw.split("\n").length;
173
+ const newToken = this.blockquote(newText)!;
174
+ tokens[tokens.length - 1] = newToken;
175
+
176
+ raw = raw.substring(0, raw.length - oldToken.raw.length) + newToken.raw;
177
+ text = text.substring(0, text.length - oldToken.text.length) + newToken.text;
178
+ break;
179
+ } else if (lastToken?.type === "list") {
180
+ // include continuation in nested list
181
+ const oldToken = lastToken;
182
+ const newText = oldToken.raw + "\n" + lines.join("\n");
183
+ // re-lexing the nested list restarts at its first source line
184
+ this.lexer.line = startLine + raw.split("\n").length - oldToken.raw.split("\n").length;
185
+ const newToken = this.list(newText)!;
186
+ tokens[tokens.length - 1] = newToken;
187
+
188
+ raw = raw.substring(0, raw.length - lastToken.raw.length) + newToken.raw;
189
+ text = text.substring(0, text.length - oldToken.raw.length) + newToken.raw;
190
+ lines = newText.substring(tokens[tokens.length - 1]!.raw.length).split("\n");
191
+ continue;
192
+ }
193
+ }
194
+
195
+ // leave the counter on the blockquote's last line; `raw` has no trailing
196
+ // newline, so the pending line terminator is counted by the next space token
197
+ this.lexer.line = startLine + raw.split("\n").length - 1;
116
198
 
117
199
  const blockquoteToken: Tokens["Blockquote"] = {
118
200
  type: "blockquote",
119
- raw: cap[0],
201
+ raw,
120
202
  tokens,
121
203
  text,
122
204
  };
@@ -169,13 +251,13 @@ export class Tokenizer {
169
251
  bull = isordered ? `\\d{1,9}\\${bull.slice(-1)}` : `\\${bull}`;
170
252
 
171
253
  // Get next list item
172
- const itemRegex = new RegExp(`^( {0,3}${bull})((?:[\t ][^\\n]*)?(?:\\n|$))`);
173
- let raw = "";
174
- let itemContents = "";
254
+ const itemRegex = other.listItemRegex(bull);
175
255
  let endsWithBlankLine = false;
176
256
  // Check if current bullet point can start a new List Item
177
257
  while (src) {
178
258
  let endEarly = false;
259
+ let raw = "";
260
+ let itemContents = "";
179
261
  if (!(cap = itemRegex.exec(src))) {
180
262
  break;
181
263
  }
@@ -188,20 +270,22 @@ export class Tokenizer {
188
270
  raw = cap[0];
189
271
  src = src.substring(raw.length);
190
272
 
191
- let line = cap[2]!
192
- .split("\n", 1)[0]!
193
- .replace(/^\t+/, (t: string) => " ".repeat(3 * t.length));
273
+ let line = expandTabs(cap[2]!.split("\n", 1)[0]!, cap[1]!.length);
194
274
  let nextLine = src.split("\n", 1)[0] ?? "";
195
275
 
196
- let indent = 0;
197
- indent = cap[2]!.search(/[^ ]/); // Find first non-space char
198
- indent = indent > 4 ? 1 : indent; // Treat indented code blocks (> 4 spaces) as having only 1 indent
199
- itemContents = line.slice(indent);
200
- indent += cap[1]!.length;
276
+ let blankLine = !line.trim();
201
277
 
202
- let blankLine = false;
278
+ let indent = 0;
279
+ if (blankLine) {
280
+ indent = cap[1]!.length + 1;
281
+ } else {
282
+ indent = line.search(other.nonSpaceChar); // Find first non-space char
283
+ indent = indent > 4 ? 1 : indent; // Treat indented code blocks (> 4 spaces) as having only 1 indent
284
+ itemContents = line.slice(indent);
285
+ indent += cap[1]!.length;
286
+ }
203
287
 
204
- if (!line && /^ *$/.test(nextLine)) {
288
+ if (blankLine && other.blankLine.test(nextLine)) {
205
289
  // Items begin with at most one blank line
206
290
  raw += nextLine + "\n";
207
291
  src = src.substring(nextLine.length + 1);
@@ -209,19 +293,18 @@ export class Tokenizer {
209
293
  }
210
294
 
211
295
  if (!endEarly) {
212
- const nextBulletRegex = new RegExp(
213
- `^ {0,${Math.min(3, indent - 1)}}(?:[*+-]|\\d{1,9}[.)])((?:[ \t][^\\n]*)?(?:\\n|$))`,
214
- );
215
- const hrRegex = new RegExp(
216
- `^ {0,${Math.min(3, indent - 1)}}((?:- *){3,}|(?:_ *){3,}|(?:\\* *){3,})(?:\\n+|$)`,
217
- );
218
- const fencesBeginRegex = new RegExp(`^ {0,${Math.min(3, indent - 1)}}(?:\`\`\`|~~~)`);
219
- const headingBeginRegex = new RegExp(`^ {0,${Math.min(3, indent - 1)}}#`);
296
+ const nextBulletRegex = other.nextBulletRegex(indent);
297
+ const hrRegex = other.hrRegex(indent);
298
+ const fencesBeginRegex = other.fencesBeginRegex(indent);
299
+ const headingBeginRegex = other.headingBeginRegex(indent);
300
+ const htmlBeginRegex = other.htmlBeginRegex(indent);
301
+ const blockquoteBeginRegex = other.blockquoteBeginRegex(indent);
220
302
 
221
303
  // Check if following lines should be included in List Item
222
304
  while (src) {
223
305
  const rawLine = src.split("\n", 1)[0] ?? "";
224
306
  nextLine = rawLine;
307
+ const nextLineWithoutTabs = nextLine.replace(other.tabCharGlobal, " ");
225
308
 
226
309
  // End list item if found code fences
227
310
  if (fencesBeginRegex.test(nextLine)) {
@@ -233,19 +316,29 @@ export class Tokenizer {
233
316
  break;
234
317
  }
235
318
 
319
+ // End list item if found start of html block
320
+ if (htmlBeginRegex.test(nextLine)) {
321
+ break;
322
+ }
323
+
324
+ // End list item if found start of blockquote
325
+ if (blockquoteBeginRegex.test(nextLine)) {
326
+ break;
327
+ }
328
+
236
329
  // End list item if found start of new bullet
237
330
  if (nextBulletRegex.test(nextLine)) {
238
331
  break;
239
332
  }
240
333
 
241
334
  // Horizontal rule found
242
- if (hrRegex.test(src)) {
335
+ if (hrRegex.test(nextLine)) {
243
336
  break;
244
337
  }
245
338
 
246
- if (nextLine.search(/[^ ]/) >= indent || !nextLine.trim()) {
339
+ if (nextLineWithoutTabs.search(other.nonSpaceChar) >= indent || !nextLine.trim()) {
247
340
  // Dedent if possible
248
- itemContents += "\n" + nextLine.slice(indent);
341
+ itemContents += "\n" + nextLineWithoutTabs.slice(indent);
249
342
  } else {
250
343
  // not enough indentation
251
344
  if (blankLine) {
@@ -253,7 +346,7 @@ export class Tokenizer {
253
346
  }
254
347
 
255
348
  // paragraph continuation unless last line was a different block level element
256
- if (line.search(/[^ ]/) >= 4) {
349
+ if (line.replace(other.tabCharGlobal, " ").search(other.nonSpaceChar) >= 4) {
257
350
  // indented code block
258
351
  break;
259
352
  }
@@ -270,14 +363,12 @@ export class Tokenizer {
270
363
  itemContents += "\n" + nextLine;
271
364
  }
272
365
 
273
- if (!blankLine && !nextLine.trim()) {
274
- // Check if current line is blank
275
- blankLine = true;
276
- }
366
+ // Check if current line is blank
367
+ blankLine = !nextLine.trim();
277
368
 
278
369
  raw += rawLine + "\n";
279
370
  src = src.substring(rawLine.length + 1);
280
- line = nextLine.slice(indent);
371
+ line = nextLineWithoutTabs.slice(indent);
281
372
  }
282
373
  }
283
374
 
@@ -285,25 +376,15 @@ export class Tokenizer {
285
376
  // If the previous item ended with a blank line, the list is loose
286
377
  if (endsWithBlankLine) {
287
378
  list.loose = true;
288
- } else if (/\n *\n *$/.test(raw)) {
379
+ } else if (other.doubleBlankLine.test(raw)) {
289
380
  endsWithBlankLine = true;
290
381
  }
291
382
  }
292
383
 
293
- let istask: RegExpExecArray | null = null;
294
- let ischecked: boolean | undefined;
295
- // Check for task list items
296
- istask = /^\[[ xX]\] /.exec(itemContents);
297
- if (istask) {
298
- ischecked = istask[0] !== "[ ] ";
299
- itemContents = itemContents.replace(/^\[[ xX]\] +/, "");
300
- }
301
-
302
384
  list.items.push({
303
385
  type: "list_item",
304
386
  raw,
305
- task: Boolean(istask),
306
- checked: ischecked,
387
+ task: other.listIsTask.test(itemContents),
307
388
  loose: false,
308
389
  text: itemContents,
309
390
  tokens: [],
@@ -313,37 +394,103 @@ export class Tokenizer {
313
394
  list.raw += raw;
314
395
  }
315
396
 
397
+ const lastItem = list.items[list.items.length - 1];
398
+ if (!lastItem) {
399
+ // not a list since there were no items
400
+ return undefined;
401
+ }
402
+
316
403
  // Do not consume newlines at end of final item. Alternatively, make itemRegex *start* with any newlines to simplify/speed up endsWithBlankLine logic
317
- const lastTrimmed = raw.trimEnd();
404
+ const lastTrimmed = lastItem.raw.trimEnd();
318
405
 
319
- if (list.items[list.items.length - 1]!.sourceMap) {
320
- this.lexer.line -= raw.length - lastTrimmed.length;
406
+ if (lastItem.sourceMap) {
407
+ // give back the trimmed trailing newlines; the counter tracks lines, not chars
408
+ this.lexer.line -= (lastItem.raw.slice(lastTrimmed.length).match(/\n/g) ?? []).length;
321
409
  }
322
410
 
323
- list.items[list.items.length - 1]!.raw = lastTrimmed;
324
- list.items[list.items.length - 1]!.text = itemContents.trimEnd();
411
+ lastItem.raw = lastTrimmed;
412
+ lastItem.text = lastItem.text.trimEnd();
325
413
  list.raw = list.raw.trimEnd();
326
414
 
327
415
  // Item child tokens handled here at end because we needed to have the final item to trim it first
328
- for (let i = 0, listItemsLen = list.items.length; i < listItemsLen; i++) {
416
+ // save/restore top: blockTokens resets it to true on exit, and a nested list
417
+ // lexed with a stale top=true would advance the sourcemap line counter
418
+ const top = this.lexer.state.top;
419
+ for (const item of list.items) {
329
420
  this.lexer.state.top = false;
330
- list.items[i]!.tokens = this.lexer.blockTokens(list.items[i]!.text, []);
421
+ item.tokens = this.lexer.blockTokens(item.text, []);
422
+
423
+ const itemToken = item.tokens[0];
424
+ if (item.task && (itemToken?.type === "text" || itemToken?.type === "paragraph")) {
425
+ // Remove checkbox markdown from item tokens
426
+ item.text = item.text.replace(other.listReplaceTask, "");
427
+ itemToken.raw = itemToken.raw.replace(other.listReplaceTask, "");
428
+ itemToken.text = itemToken.text.replace(other.listReplaceTask, "");
429
+ for (let i = this.lexer.inlineQueue.length - 1; i >= 0; i--) {
430
+ if (other.listIsTask.test(this.lexer.inlineQueue[i]!.src)) {
431
+ this.lexer.inlineQueue[i]!.src = this.lexer.inlineQueue[i]!.src.replace(
432
+ other.listReplaceTask,
433
+ "",
434
+ );
435
+ break;
436
+ }
437
+ }
438
+
439
+ const taskRaw = other.listTaskCheckbox.exec(item.raw);
440
+ if (taskRaw) {
441
+ const checkboxToken: Tokens["Checkbox"] = {
442
+ type: "checkbox",
443
+ raw: taskRaw[0] + " ",
444
+ checked: taskRaw[0] !== "[ ]",
445
+ };
446
+ item.checked = checkboxToken.checked;
447
+ const firstToken = item.tokens[0];
448
+ if (list.loose) {
449
+ if (
450
+ firstToken &&
451
+ (firstToken.type === "paragraph" || firstToken.type === "text") &&
452
+ "tokens" in firstToken
453
+ ) {
454
+ firstToken.raw = checkboxToken.raw + firstToken.raw;
455
+ firstToken.text = checkboxToken.raw + firstToken.text;
456
+ firstToken.tokens.unshift(checkboxToken);
457
+ } else {
458
+ item.tokens.unshift({
459
+ type: "paragraph",
460
+ raw: checkboxToken.raw,
461
+ text: checkboxToken.raw,
462
+ tokens: [checkboxToken],
463
+ sourceMap: undefined,
464
+ });
465
+ }
466
+ } else {
467
+ item.tokens.unshift(checkboxToken);
468
+ }
469
+ }
470
+ } else if (item.task) {
471
+ item.task = false;
472
+ }
331
473
 
332
474
  if (!list.loose) {
333
475
  // Check if list should be loose
334
- const spacers = list.items[i]!.tokens.filter((t) => t.type === "space");
476
+ const spacers = item.tokens.filter((t) => t.type === "space");
335
477
  const hasMultipleLineBreaks =
336
- // eslint-disable-next-line
337
- spacers.length > 0 && spacers.some((t: any) => /\n.*\n/.test(t.raw));
478
+ spacers.length > 0 && spacers.some((t) => other.anyLine.test(t.raw));
338
479
 
339
480
  list.loose = hasMultipleLineBreaks;
340
481
  }
341
482
  }
483
+ this.lexer.state.top = top;
342
484
 
343
485
  // Set all items to loose if list is loose
344
486
  if (list.loose) {
345
- for (let i = 0, listItemsLen = list.items.length; i < listItemsLen; i++) {
346
- list.items[i]!.loose = true;
487
+ for (const item of list.items) {
488
+ item.loose = true;
489
+ for (const token of item.tokens) {
490
+ if (token.type === "text") {
491
+ (token as unknown as Tokens["Paragraph"]).type = "paragraph";
492
+ }
493
+ }
347
494
  }
348
495
  }
349
496
 
@@ -380,13 +527,14 @@ export class Tokenizer {
380
527
  const cap = block.html.exec(src);
381
528
  if (!cap) return undefined;
382
529
 
530
+ const raw = trimTrailingBlankLines(cap[0]);
383
531
  const token: Tokens["HTML"] = {
384
532
  type: "html",
385
533
  block: true,
386
- raw: cap[0],
534
+ raw,
387
535
  pre: cap[1] === "pre" || cap[1] === "script" || cap[1] === "style",
388
- text: cap[0],
389
- sourceMap: this.lexer.getSourceMap(cap[0]),
536
+ text: raw,
537
+ sourceMap: this.lexer.getSourceMap(raw),
390
538
  };
391
539
 
392
540
  /*
@@ -454,20 +602,21 @@ export class Tokenizer {
454
602
  const cap = block.def.exec(src);
455
603
  if (!cap) return undefined;
456
604
 
457
- const tag = cap[1]!.toLowerCase().replace(/\s+/g, " ");
605
+ const tag = cap[1]!.toLowerCase().replace(other.multipleSpaceGlobal, " ");
458
606
  const href = cap[2]
459
- ? cap[2].replace(/^<(.*)>$/, "$1").replace(inline.anyPunctuation, "$1")
607
+ ? cap[2].replace(other.hrefBrackets, "$1").replace(inline.anyPunctuation, "$1")
460
608
  : "";
461
609
  const title = cap[3]
462
610
  ? cap[3].substring(1, cap[3].length - 1).replace(inline.anyPunctuation, "$1")
463
611
  : "";
612
+ const raw = rtrim(cap[0], "\n");
464
613
  return {
465
614
  type: "def",
466
615
  tag,
467
- raw: cap[0],
616
+ raw,
468
617
  href,
469
618
  title,
470
- sourceMap: this.lexer.getSourceMap(cap[0]),
619
+ sourceMap: this.lexer.getSourceMap(raw),
471
620
  };
472
621
  }
473
622
 
@@ -475,76 +624,62 @@ export class Tokenizer {
475
624
  const cap = block.table.exec(src);
476
625
  if (!cap?.[2]) return;
477
626
 
478
- if (!/[:|]/.test(cap[2])) {
627
+ if (!other.tableDelimiter.test(cap[2])) {
479
628
  // delimiter row must have a pipe (|) or colon (:) otherwise it is a setext heading
480
629
  return;
481
630
  }
482
631
 
632
+ const headers = splitCells(cap[1]!);
633
+ const aligns = cap[2].replace(other.tableAlignChars, "").split("|");
634
+ const rows = cap[3]?.trim() ? cap[3].replace(other.tableRowBlankLine, "").split("\n") : [];
635
+
636
+ if (headers.length !== aligns.length) return;
637
+
483
638
  const item: Tokens["Table"] = {
484
639
  type: "table",
485
- raw: cap[0],
486
- header: splitCells(cap[1]!).map((c) => ({
487
- type: "tablecell",
488
- raw: c,
489
- text: c,
490
- tokens: [],
491
- })),
640
+ raw: rtrim(cap[0], "\n"),
641
+ header: [],
492
642
  align: [],
493
643
  rows: [],
494
- sourceMap: this.lexer.getSourceMap(cap[0]),
644
+ sourceMap: this.lexer.getSourceMap(rtrim(cap[0], "\n")),
495
645
  };
496
646
 
497
- const align = cap[2].replace(/^\||\| *$/g, "").split("|") as (string | null)[];
498
- const rows = cap[3]?.trim() ? cap[3].replace(/\n[ \t]*$/, "").split("\n") : [];
499
-
500
- if (item.header.length !== align.length) return;
501
-
502
- let l = align.length;
503
- let i, j, k, row;
504
- for (i = 0; i < l; i++) {
505
- const alignStr = align[i];
506
- if (alignStr) {
507
- if (/^ *-+: *$/.test(alignStr)) {
508
- item.align.push("right");
509
- } else if (/^ *:-+: *$/.test(alignStr)) {
510
- item.align.push("center");
511
- } else if (/^ *:-+ *$/.test(alignStr)) {
512
- item.align.push("left");
513
- } else {
514
- item.align.push(null);
515
- }
647
+ for (const align of aligns) {
648
+ if (other.tableAlignRight.test(align)) {
649
+ item.align.push("right");
650
+ } else if (other.tableAlignCenter.test(align)) {
651
+ item.align.push("center");
652
+ } else if (other.tableAlignLeft.test(align)) {
653
+ item.align.push("left");
654
+ } else {
655
+ item.align.push(null);
516
656
  }
517
657
  }
518
658
 
519
- l = rows.length;
520
- for (i = 0; i < l; i++) {
659
+ for (let i = 0, len = headers.length; i < len; i++) {
660
+ item.header.push({
661
+ type: "tablecell",
662
+ raw: headers[i]!,
663
+ text: headers[i]!,
664
+ tokens: this.lexer.inline(headers[i]!),
665
+ header: true,
666
+ align: item.align[i]!,
667
+ });
668
+ }
669
+
670
+ for (const row of rows) {
521
671
  item.rows.push(
522
- splitCells(rows[i] as unknown as string, item.header.length).map((c) => ({
523
- type: "tablecell",
524
- raw: c,
525
- text: c,
526
- tokens: [],
672
+ splitCells(row, headers.length).map((cell, i) => ({
673
+ type: "tablecell" as const,
674
+ raw: cell,
675
+ text: cell,
676
+ tokens: this.lexer.inline(cell),
677
+ header: false,
678
+ align: item.align[i]!,
527
679
  })),
528
680
  );
529
681
  }
530
682
 
531
- // parse child tokens inside headers and cells
532
-
533
- // header child tokens
534
- l = item.header.length;
535
- for (j = 0; j < l; j++) {
536
- item.header[j]!.tokens = this.lexer.inline(item.header[j]!.text);
537
- }
538
-
539
- // cell child tokens
540
- l = item.rows.length;
541
- for (j = 0; j < l; j++) {
542
- row = item.rows[j]!;
543
- for (k = 0; k < row.length; k++) {
544
- row[k]!.tokens = this.lexer.inline(row[k]!.text);
545
- }
546
- }
547
-
548
683
  return item;
549
684
  }
550
685
 
@@ -552,13 +687,15 @@ export class Tokenizer {
552
687
  const cap = block.lheading.exec(src);
553
688
  if (!cap) return undefined;
554
689
 
690
+ const text = cap[1]!.trim();
691
+ const raw = rtrim(cap[0], "\n");
555
692
  return {
556
693
  type: "heading",
557
- raw: cap[0],
694
+ raw,
558
695
  depth: cap[2]!.startsWith("=") ? 1 : 2,
559
- text: cap[1]!,
560
- tokens: this.lexer.inline(cap[1]!),
561
- sourceMap: this.lexer.getSourceMap(cap[0]),
696
+ text,
697
+ tokens: this.lexer.inline(text),
698
+ sourceMap: this.lexer.getSourceMap(raw),
562
699
  };
563
700
  }
564
701
 
@@ -596,7 +733,7 @@ export class Tokenizer {
596
733
  return {
597
734
  type: "escape",
598
735
  raw: cap[0],
599
- text: escape(cap[1]!),
736
+ text: cap[1]!,
600
737
  };
601
738
  }
602
739
 
@@ -604,14 +741,14 @@ export class Tokenizer {
604
741
  const cap = inline.tag.exec(src);
605
742
  if (!cap) return undefined;
606
743
 
607
- if (!this.lexer.state.inLink && /^<a /i.test(cap[0])) {
744
+ if (!this.lexer.state.inLink && other.startATag.test(cap[0])) {
608
745
  this.lexer.state.inLink = true;
609
- } else if (this.lexer.state.inLink && /^<\/a>/i.test(cap[0])) {
746
+ } else if (this.lexer.state.inLink && other.endATag.test(cap[0])) {
610
747
  this.lexer.state.inLink = false;
611
748
  }
612
- if (!this.lexer.state.inRawBlock && /^<(pre|code|kbd|script)(\s|>)/i.test(cap[0])) {
749
+ if (!this.lexer.state.inRawBlock && other.startPreScriptTag.test(cap[0])) {
613
750
  this.lexer.state.inRawBlock = true;
614
- } else if (this.lexer.state.inRawBlock && /^<\/(pre|code|kbd|script)(\s|>)/i.test(cap[0])) {
751
+ } else if (this.lexer.state.inRawBlock && other.endPreScriptTag.test(cap[0])) {
615
752
  this.lexer.state.inRawBlock = false;
616
753
  }
617
754
 
@@ -644,6 +781,10 @@ export class Tokenizer {
644
781
  } else {
645
782
  // find closing parenthesis
646
783
  const lastParenIndex = findClosingBracket(cap[2]!, "()");
784
+ if (lastParenIndex === -2) {
785
+ // more open parens than closed
786
+ return;
787
+ }
647
788
  if (lastParenIndex > -1) {
648
789
  const start = cap[0].startsWith("!") ? 5 : 4;
649
790
  const linkLen = start + cap[1]!.length + lastParenIndex;
@@ -699,17 +840,16 @@ export class Tokenizer {
699
840
  ): Tokens["Em"] | Tokens["Strong"] | undefined {
700
841
  let match = inline.emStrong.lDelim.exec(src);
701
842
  if (!match) return;
843
+ if (!match[1] && !match[2] && !match[3] && !match[4]) return;
702
844
 
703
845
  // _ can't be between two alphanumerics. \p{L}\p{N} includes non-english alphabet/numbers as well
704
- if (match[3] && /[\p{L}\p{N}]/u.exec(prevChar)) return;
846
+ if (match[4] && other.unicodeAlphaNumeric.exec(prevChar)) return;
705
847
 
706
- // eslint-disable-next-line
707
- const nextChar = match[1] || match[2] || "";
848
+ const nextChar = match[1] || match[3] || "";
708
849
 
709
850
  if (!nextChar || !prevChar || inline.punctuation.exec(prevChar)) {
710
- // unicode Regex counts emoji as 1 char; spread into array for proper count (used multiple times below)
711
- // eslint-disable-next-line @typescript-eslint/no-misused-spread
712
- const lLength = [...match[0]].length - 1;
851
+ // unicode Regex counts emoji as 1 char; convert to array for proper count (used multiple times below)
852
+ const lLength = Array.from(match[0]).length - 1;
713
853
  let rDelim,
714
854
  rLength,
715
855
  delimTotal = lLength,
@@ -724,13 +864,11 @@ export class Tokenizer {
724
864
  maskedSrc = maskedSrc.slice(-1 * src.length + lLength);
725
865
 
726
866
  while ((match = endReg.exec(maskedSrc)) != null) {
727
- // eslint-disable-next-line
728
867
  rDelim = match[1] || match[2] || match[3] || match[4] || match[5] || match[6];
729
868
 
730
869
  if (!rDelim) continue; // skip single * in __abc*abc__
731
870
 
732
- // eslint-disable-next-line @typescript-eslint/no-misused-spread
733
- rLength = [...rDelim].length;
871
+ rLength = Array.from(rDelim).length;
734
872
 
735
873
  if (match[3] || match[4]) {
736
874
  // found another Left Delim
@@ -751,8 +889,7 @@ export class Tokenizer {
751
889
  // Remove extra characters. *a*** -> *a*
752
890
  rLength = Math.min(rLength, rLength + delimTotal + midDelimTotal);
753
891
  // char length can be >1 for unicode characters;
754
- // eslint-disable-next-line @typescript-eslint/no-misused-spread
755
- const lastCharLength = [...match[0]][0]!.length;
892
+ const lastCharLength = Array.from(match[0])[0]!.length;
756
893
  const raw = src.slice(0, lLength + match.index + lastCharLength + rLength);
757
894
 
758
895
  // Create `em` if smallest delimiter has odd char count. *a***
@@ -795,13 +932,12 @@ export class Tokenizer {
795
932
  const cap = inline.code.exec(src);
796
933
  if (!cap) return undefined;
797
934
 
798
- let text = cap[2]!.replace(/\n/g, " ");
799
- const hasNonSpaceChars = /[^ ]/.test(text);
935
+ let text = cap[2]!.replace(other.newLineCharGlobal, " ");
936
+ const hasNonSpaceChars = other.nonSpaceChar.test(text);
800
937
  const hasSpaceCharsOnBothEnds = text.startsWith(" ") && text.endsWith(" ");
801
938
  if (hasNonSpaceChars && hasSpaceCharsOnBothEnds) {
802
939
  text = text.substring(1, text.length - 1);
803
940
  }
804
- text = escape(text, true);
805
941
  return {
806
942
  type: "codespan",
807
943
  raw: cap[0],
@@ -819,16 +955,62 @@ export class Tokenizer {
819
955
  };
820
956
  }
821
957
 
822
- del(src: string): Tokens["Del"] | undefined {
823
- const cap = inline.del.exec(src);
824
- if (!cap) return undefined;
958
+ del(src: string, maskedSrc: string, prevChar = ""): Tokens["Del"] | undefined {
959
+ let match = inline.delLDelim.exec(src);
960
+ if (!match) return;
825
961
 
826
- return {
827
- type: "del",
828
- raw: cap[0],
829
- text: cap[2]!,
830
- tokens: this.lexer.inlineTokens(cap[2]!),
831
- };
962
+ const nextChar = match[1] || "";
963
+
964
+ if (!nextChar || !prevChar || inline.punctuation.exec(prevChar)) {
965
+ // unicode Regex counts emoji as 1 char; spread into array for proper count
966
+ const lLength = Array.from(match[0]).length - 1;
967
+ let rDelim,
968
+ rLength,
969
+ delimTotal = lLength;
970
+
971
+ const endReg = inline.delRDelim;
972
+ endReg.lastIndex = 0;
973
+
974
+ // Clip maskedSrc to same section of string as src
975
+ maskedSrc = maskedSrc.slice(-1 * src.length + lLength);
976
+
977
+ while ((match = endReg.exec(maskedSrc)) !== null) {
978
+ rDelim = match[1] || match[2] || match[3] || match[4] || match[5] || match[6];
979
+
980
+ if (!rDelim) continue;
981
+
982
+ rLength = Array.from(rDelim).length;
983
+
984
+ if (rLength !== lLength) continue;
985
+
986
+ if (match[3] || match[4]) {
987
+ // found another Left Delim
988
+ delimTotal += rLength;
989
+ continue;
990
+ }
991
+
992
+ delimTotal -= rLength;
993
+
994
+ if (delimTotal > 0) continue; // Haven't found enough closing delimiters
995
+
996
+ // Remove extra characters
997
+ rLength = Math.min(rLength, rLength + delimTotal);
998
+ // char length can be >1 for unicode characters
999
+ const lastCharLength = Array.from(match[0])[0]!.length;
1000
+ const raw = src.slice(0, lLength + match.index + lastCharLength + rLength);
1001
+
1002
+ // Create del token - only single ~ or double ~~ supported
1003
+ const text = raw.slice(lLength, -lLength);
1004
+ return {
1005
+ type: "del",
1006
+ raw,
1007
+ text,
1008
+ tokens: this.lexer.inlineTokens(text),
1009
+ };
1010
+ }
1011
+ }
1012
+
1013
+ return undefined;
832
1014
  }
833
1015
 
834
1016
  autolink(src: string): Tokens["Link"] | undefined {
@@ -837,10 +1019,10 @@ export class Tokenizer {
837
1019
 
838
1020
  let text, href;
839
1021
  if (cap[2] === "@") {
840
- text = escape(cap[1]!);
1022
+ text = cap[1]!;
841
1023
  href = "mailto:" + text;
842
1024
  } else {
843
- text = escape(cap[1]!);
1025
+ text = cap[1]!;
844
1026
  href = text;
845
1027
  }
846
1028
 
@@ -865,7 +1047,7 @@ export class Tokenizer {
865
1047
  if ((cap = inline.url.exec(src))) {
866
1048
  let text, href;
867
1049
  if (cap[2] === "@") {
868
- text = escape(cap[0]);
1050
+ text = cap[0];
869
1051
  href = "mailto:" + text;
870
1052
  } else {
871
1053
  // do extended autolink path validation
@@ -874,7 +1056,7 @@ export class Tokenizer {
874
1056
  prevCapZero = cap[0];
875
1057
  cap[0] = inline.backpedal.exec(cap[0])![0];
876
1058
  } while (prevCapZero !== cap[0]);
877
- text = escape(cap[0]);
1059
+ text = cap[0];
878
1060
  if (cap[1] === "www.") {
879
1061
  href = "http://" + cap[0];
880
1062
  } else {
@@ -903,16 +1085,11 @@ export class Tokenizer {
903
1085
  const cap = inline.text.exec(src);
904
1086
  if (!cap) return undefined;
905
1087
 
906
- let text;
907
- if (this.lexer.state.inRawBlock) {
908
- text = cap[0];
909
- } else {
910
- text = escape(cap[0]);
911
- }
912
1088
  return {
913
1089
  type: "text",
914
1090
  raw: cap[0],
915
- text,
1091
+ text: cap[0],
1092
+ escaped: this.lexer.state.inRawBlock,
916
1093
  };
917
1094
  }
918
1095