pantsdown 2.2.7 → 2.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -11,7 +11,8 @@ type InlineRuleNames =
11
11
  | "reflinkSearch"
12
12
  | "code"
13
13
  | "br"
14
- | "del"
14
+ | "delLDelim"
15
+ | "delRDelim"
15
16
  | "url"
16
17
  | "text"
17
18
  | "emStrong"
@@ -24,45 +25,77 @@ type InlineRuleNames =
24
25
 
25
26
  // list of unicode punctuation marks, plus any missing characters from CommonMark spec
26
27
  const punctuation = "\\p{P}\\p{S}";
28
+ const _punctuation = /[\p{P}\p{S}]/u;
29
+ const _punctuationOrSpace = /[\s\p{P}\p{S}]/u;
30
+ const _notPunctuationOrSpace = /[^\s\p{P}\p{S}]/u;
31
+
32
+ // GFM allows ~ inside strong and em for strikethrough
33
+ const _punctuationGfmStrongEm = /(?!~)[\p{P}\p{S}]/u;
34
+ const _punctuationOrSpaceGfmStrongEm = /(?!~)[\s\p{P}\p{S}]/u;
35
+ const _notPunctuationOrSpaceGfmStrongEm = /(?:[^\s\p{P}\p{S}]|~)/u;
36
+
27
37
  const title = /"(?:\\"?|[^"\\])*"|'(?:\\'?|[^'\\])*'|\((?:\\\)?|[^)\\])*\)/;
28
- const href = /<(?:\\.|[^\n<>\\])+>|[^\s\x00-\x1f]*/;
38
+ const href = /<(?:\\.|[^\n<>\\])+>|[^ \t\n\x00-\x1f]+|(?=\))/;
29
39
  const scheme = /[a-zA-Z][a-zA-Z0-9+.-]{1,31}/;
30
40
  const comment = edit(blockComment).replace("(?:-->|$)", "-->").getRegex();
31
41
  const attribute = /\s+[a-zA-Z:_][\w.:-]*(?:\s*=\s*"[^"]*"|\s*=\s*'[^']*'|\s*=\s*[^\s"'=<>`]+)?/;
32
- const label = /(?:\[(?:\\.|[^\[\]\\])*\]|\\.|`[^`]*`|[^\[\]\\`])*?/;
42
+ // codespan branches carry the #3918 ReDoS fix (`+(?!`) head, ``+(?=\]) tail)
43
+ const label =
44
+ /(?:\[(?:\\[\s\S]|[^\[\]\\])*\]|\\[\s\S]|`+(?!`)[^`]*?`+(?!`)|``+(?=\])|[^\[\]\\`])*?/;
33
45
  const email =
34
46
  /[a-zA-Z0-9.!#$%&'*+/=?^_`{|}~-]+(@)[a-zA-Z0-9](?:[a-zA-Z0-9-]{0,61}[a-zA-Z0-9])?(?:\.[a-zA-Z0-9](?:[a-zA-Z0-9-]{0,61}[a-zA-Z0-9])?)+(?![-_])/;
35
47
  const extended_email = /[A-Za-z0-9._+-]+(@)[a-zA-Z0-9-_]+(?:\.[a-zA-Z0-9-_]*[a-zA-Z0-9])+(?![-_])/;
36
48
 
37
- const inline_punctuation = edit(/^((?![*_])[\spunctuation])/, "u")
38
- .replace(/punctuation/g, punctuation)
49
+ const inline_punctuation = edit(/^((?![*_])punctSpace)/, "u")
50
+ .replace(/punctSpace/g, _punctuationOrSpace)
39
51
  .getRegex();
40
52
 
41
53
  // sequences em should skip over [title](link), `code`, <html>
42
- const inline_blockSkip = /\[[^[\]]*?\]\([^\(\)]*?\)|`[^`]*?`|<[^<>]*?>/g;
54
+ // upstream builds this via a codePattern placeholder and a lookbehind fallback;
55
+ // bun (JSC) supports lookbehind so we inline the lookbehind form directly
56
+ const inline_blockSkip =
57
+ /\[(?:[^\[\]`]|(?<a>`+)[^`]+\k<a>(?!`))*?\]\((?:\\[\s\S]|[^\\\(\)]|\((?:\\[\s\S]|[^\\\(\)])*\))*\)|(?<!`)(?<b>`+)[^`]+\k<b>(?!`)|<(?! )[^<>]*?>/g;
58
+
59
+ const emStrongLDelimCore = /^(?:\*+(?:((?!\*)punct)|([^\s*]))?)|^_+(?:((?!_)punct)|([^\s_]))?/;
60
+
61
+ const emStrongRDelimAstCore =
62
+ "^[^_*]*?__[^_*]*?\\*[^_*]*?(?=__)" + // Skip orphan inside strong
63
+ "|[^*]+(?=[^*])" + // Consume to delim
64
+ "|(?!\\*)punct(\\*+)(?=[\\s]|$)" + // (1) #*** can only be a Right Delimiter
65
+ "|notPunctSpace(\\*+)(?!\\*)(?=punctSpace|$)" + // (2) a***#, a*** can only be a Right Delimiter
66
+ "|(?!\\*)punctSpace(\\*+)(?=notPunctSpace)" + // (3) #***a, ***a can only be Left Delimiter
67
+ "|[\\s](\\*+)(?!\\*)(?=punct)" + // (4) ***# can only be Left Delimiter
68
+ "|(?!\\*)punct(\\*+)(?!\\*)(?=punct)" + // (5) #***# can be either Left or Right Delimiter
69
+ "|notPunctSpace(\\*+)(?=notPunctSpace)"; // (6) a***a can be either Left or Right Delimiter
43
70
 
44
71
  const inline_emStrong = {
45
- lDelim: edit(/^(?:\*+(?:((?!\*)[punct])|[^\s*]))|^_+(?:((?!_)[punct])|([^\s_]))/, "u")
46
- .replace(/punct/g, punctuation)
47
- .getRegex(),
48
- // (1) and (2) can only be a Right Delimiter. (3) and (4) can only be Left. (5) and (6) can be either Left or Right.
49
- // | Skip orphan inside strong | Consume to delim | (1) #*** | (2) a***#, a*** | (3) #***a, ***a | (4) ***# | (5) #***# | (6) a***a
50
- rDelimAst: edit(
51
- /^[^_*]*?__[^_*]*?\*[^_*]*?(?=__)|[^*]+(?=[^*])|(?!\*)[punct](\*+)(?=[\s]|$)|[^punct\s](\*+)(?!\*)(?=[punct\s]|$)|(?!\*)[punct\s](\*+)(?=[^punct\s])|[\s](\*+)(?!\*)(?=[punct])|(?!\*)[punct](\*+)(?!\*)(?=[punct])|[^punct\s](\*+)(?=[^punct\s])/,
52
- "gu",
53
- )
54
- .replace(/punct/g, punctuation)
72
+ // GFM variants: ~ is not treated as punctuation so strikethrough
73
+ // can nest directly inside strong/em (upstream emStrongLDelimGfm)
74
+ lDelim: edit(emStrongLDelimCore, "u").replace(/punct/g, _punctuationGfmStrongEm).getRegex(),
75
+ // upstream emStrongRDelimAstGfm
76
+ rDelimAst: edit(emStrongRDelimAstCore, "gu")
77
+ .replace(/notPunctSpace/g, _notPunctuationOrSpaceGfmStrongEm)
78
+ .replace(/punctSpace/g, _punctuationOrSpaceGfmStrongEm)
79
+ .replace(/punct/g, _punctuationGfmStrongEm)
55
80
  .getRegex(),
81
+ // (6) Not allowed for _
56
82
  rDelimUnd: edit(
57
- // ^- Not allowed for _
58
- /^[^_*]*?\*\*[^_*]*?_[^_*]*?(?=\*\*)|[^_]+(?=[^_])|(?!_)[punct](_+)(?=[\s]|$)|[^punct\s](_+)(?!_)(?=[punct\s]|$)|(?!_)[punct\s](_+)(?=[^punct\s])|[\s](_+)(?!_)(?=[punct])|(?!_)[punct](_+)(?!_)(?=[punct])/,
83
+ "^[^_*]*?\\*\\*[^_*]*?_[^_*]*?(?=\\*\\*)" + // Skip orphan inside strong
84
+ "|[^_]+(?=[^_])" + // Consume to delim
85
+ "|(?!_)punct(_+)(?=[\\s]|$)" + // (1) #___ can only be a Right Delimiter
86
+ "|notPunctSpace(_+)(?!_)(?=punctSpace|$)" + // (2) a___#, a___ can only be a Right Delimiter
87
+ "|(?!_)punctSpace(_+)(?=notPunctSpace)" + // (3) #___a, ___a can only be Left Delimiter
88
+ "|[\\s](_+)(?!_)(?=punct)" + // (4) ___# can only be Left Delimiter
89
+ "|(?!_)punct(_+)(?!_)(?=punct)", // (5) #___# can be either Left or Right Delimiter
59
90
  "gu",
60
91
  )
61
- .replace(/punct/g, punctuation)
92
+ .replace(/notPunctSpace/g, _notPunctuationOrSpace)
93
+ .replace(/punctSpace/g, _punctuationOrSpace)
94
+ .replace(/punct/g, _punctuation)
62
95
  .getRegex(),
63
96
  };
64
97
 
65
- const inline_anyPunctuation = edit(/\\[punct]/g, "gu")
98
+ const inline_anyPunctuation = edit(/\\([punct])/g, "gu")
66
99
  .replace(/punct/g, punctuation)
67
100
  .getRegex();
68
101
 
@@ -83,7 +116,9 @@ const inline_tag = edit(
83
116
  .replace("attribute", attribute)
84
117
  .getRegex();
85
118
 
86
- const inline_link = edit(/^!?\[(label)\]\(\s*(href)(?:\s+(title))?\s*\)/)
119
+ const inline_link = edit(
120
+ /^!?\[(label)\]\(\s*(href)(?:(?:[ \t]+(?:\n[ \t]*)?|\n[ \t]*)(title))?\s*\)/,
121
+ )
87
122
  .replace("label", label)
88
123
  .replace("href", href)
89
124
  .replace("title", title)
@@ -103,19 +138,47 @@ const inline_reflinkSearch = edit("reflink|nolink(?!\\()", "g")
103
138
  .replace("nolink", inline_nolink)
104
139
  .getRegex();
105
140
 
106
- const inline_escape = edit(/^\\([!"#$%&'()*+,\-./:;<=>?@\[\]\\^_`{|}~])/)
107
- .replace("])", "~|])")
108
- .getRegex();
141
+ const inline_escape = /^\\([!"#$%&'()*+,\-./:;<=>?@\[\]\\^_`{|}~])/;
109
142
 
110
143
  const inline_backpedal =
111
144
  /(?:[^?!.,:;*_'"~()&]+|\([^)]*\)|&(?![a-zA-Z0-9]+;$)|[?!.,:;*_'"~)]+(?!$))+/;
112
145
 
113
- const inline_del = /^(~~?)(?=[^\s~])([\s\S]*?[^\s~])\1(?=[^~]|$)/;
146
+ // Tilde left delimiter for strikethrough (similar to emStrongLDelim for asterisk)
147
+ const inline_delLDelim = edit(/^~~?(?:((?!~)punct)|[^\s~])/, "u")
148
+ .replace(/punct/g, _punctuation)
149
+ .getRegex();
150
+
151
+ // Tilde delimiter patterns for strikethrough (similar to asterisk)
152
+ const delRDelimCore =
153
+ "^[^~]+(?=[^~])" + // Consume to delim
154
+ "|(?!~)punct(~~?)(?=[\\s]|$)" + // (1) #~~ can only be a Right Delimiter
155
+ "|notPunctSpace(~~?)(?!~)(?=punctSpace|$)" + // (2) a~~#, a~~ can only be a Right Delimiter
156
+ "|(?!~)punctSpace(~~?)(?=notPunctSpace)" + // (3) #~~a, ~~a can only be Left Delimiter
157
+ "|[\\s](~~?)(?!~)(?=punct)" + // (4) ~~# can only be Left Delimiter
158
+ "|(?!~)punct(~~?)(?!~)(?=punct)" + // (5) #~~# can be either Left or Right Delimiter
159
+ "|notPunctSpace(~~?)(?=notPunctSpace)"; // (6) a~~a can be either Left or Right Delimiter
160
+
161
+ const inline_delRDelim = edit(delRDelimCore, "gu")
162
+ .replace(/notPunctSpace/g, _notPunctuationOrSpace)
163
+ .replace(/punctSpace/g, _punctuationOrSpace)
164
+ .replace(/punct/g, _punctuation)
165
+ .getRegex();
114
166
 
115
- const inline_text =
116
- /^([`~]+|[^`~])(?:(?= {2,}\n)|(?=[a-zA-Z0-9.!#$%&'*+\/=?_`{\|}~-]+@)|[\s\S]*?(?:(?=[\\<!\[`*~_$]|\b_|https?:\/\/|ftp:\/\/|www\.|$)|[^ ](?= {2,}\n)|[^a-zA-Z0-9.!#$%&'*+\/=?_`{\|}~-](?=[a-zA-Z0-9.!#$%&'*+\/=?_`{\|}~-]+@)))/;
167
+ const inline_text = edit(
168
+ /^(`+|~+|[^`~])(?:(?=[`~])|(?= {2,}\n)|(?=[a-zA-Z0-9.!#$%&'*+\/=?_`{\|}~-]+@)|[\s\S]*?(?:(?=[\\<!\[`*~_$]|\b_|protocol:\/\/|www\.|$)|[^ ](?= {2,}\n)|[^a-zA-Z0-9.!#$%&'*+\/=?_`{\|}~-](?=[a-zA-Z0-9.!#$%&'*+\/=?_`{\|}~-]+@)))/,
169
+ )
170
+ .replace("protocol", /[hH][tT][tT][pP][sS]?|[fF][tT][pP]/)
171
+ .getRegex();
117
172
 
118
- const inline_url = edit(/^((?:ftp|https?):\/\/|www\.)(?:[a-zA-Z0-9\-]+\.?)+[^\s<]*|^email/, "i")
173
+ // DEVIATION from upstream `(?:[a-zA-Z0-9\-]+\.?)+`: the nested quantifier
174
+ // combined with the email alternative defeats JSC's start-anchor optimization
175
+ // (every exec scans the whole subject, O(n²) across the inline loop). This
176
+ // domain form matches the exact same language (fuzz-verified over 200k
177
+ // samples) — preserve it when porting upstream changes to this rule.
178
+ const inline_url = edit(
179
+ /^((?:protocol):\/\/|www\.)(?:[a-zA-Z0-9\-]+\.)*[a-zA-Z0-9\-]+\.?[^\s<]*|^email/,
180
+ )
181
+ .replace("protocol", /[fF][tT][pP]|[hH][tT][tT][pP][sS]?/)
119
182
  .replace("email", extended_email)
120
183
  .getRegex();
121
184
 
@@ -134,7 +197,8 @@ export const inline: Omit<Record<InlineRuleNames, RegExp>, "emStrong"> & {
134
197
  emStrong: inline_emStrong,
135
198
  code: /^(`+)([^`]|[^`][\s\S]*?[^`])\1(?!`)/,
136
199
  br: /^( {2,}|\\)\n(?!\s*$)/,
137
- del: inline_del,
200
+ delLDelim: inline_delLDelim,
201
+ delRDelim: inline_delRDelim,
138
202
  text: inline_text,
139
203
  punctuation: inline_punctuation,
140
204
  blockSkip: inline_blockSkip,
@@ -0,0 +1,79 @@
1
+ /**
2
+ * Regexes that don't belong to the block or inline grammars.
3
+ * Names and values mirror marked's `other` rules object (src/rules.ts,
4
+ * currently marked v18.0.7 — see "Last synced" in the root README) so
5
+ * future syncs stay diffable.
6
+ */
7
+ function cachedIndentRegex(createRegex: (indent: number) => RegExp) {
8
+ const cache: RegExp[] = [];
9
+ return (indent: number) => {
10
+ const cacheIndex = Math.max(0, Math.min(3, indent - 1));
11
+ let regex = cache[cacheIndex];
12
+ if (!regex) {
13
+ regex = createRegex(cacheIndex);
14
+ cache[cacheIndex] = regex;
15
+ }
16
+ return regex;
17
+ };
18
+ }
19
+
20
+ export const other = {
21
+ codeRemoveIndent: /^(?: {1,4}| {0,3}\t)/gm,
22
+ tabCharGlobal: /\t/g,
23
+ outputLinkReplace: /\\([\[\]])/g,
24
+ indentCodeCompensation: /^(\s+)(?:```)/,
25
+ beginningSpace: /^\s+/,
26
+ nonSpaceChar: /[^ ]/,
27
+ newLineCharGlobal: /\n/g,
28
+ multipleSpaceGlobal: /\s+/g,
29
+ blankLine: /^[ \t]*$/,
30
+ doubleBlankLine: /\n[ \t]*\n[ \t]*$/,
31
+ blockquoteSetextReplace: /\n {0,3}((?:=+|-+) *)(?=\n|$)/g,
32
+ blockquoteSetextReplace2: /^ {0,3}>[ \t]?/gm,
33
+ blockquoteStart: /^ {0,3}>/,
34
+ listIsTask: /^\[[ xX]\] +\S/,
35
+ listReplaceTask: /^\[[ xX]\] +/,
36
+ listTaskCheckbox: /\[[ xX]\]/,
37
+ anyLine: /\n.*\n/,
38
+ hrefBrackets: /^<(.*)>$/,
39
+ tableDelimiter: /[:|]/,
40
+ tableAlignChars: /^\||\| *$/g,
41
+ tableRowBlankLine: /\n[ \t]*$/,
42
+ tableAlignRight: /^ *-+: *$/,
43
+ tableAlignCenter: /^ *:-+: *$/,
44
+ tableAlignLeft: /^ *:-+ *$/,
45
+ startATag: /^<a /i,
46
+ endATag: /^<\/a>/i,
47
+ startPreScriptTag: /^<(pre|code|kbd|script)(\s|>)/i,
48
+ endPreScriptTag: /^<\/(pre|code|kbd|script)(\s|>)/i,
49
+ unicodeAlphaNumeric: /[\p{L}\p{N}]/u,
50
+ escapeTest: /[&<>"']/,
51
+ escapeReplace: /[&<>"']/g,
52
+ escapeTestNoEncode: /[<>"']|&(?!(#\d{1,7}|#[Xx][a-fA-F0-9]{1,6}|\w+);)/,
53
+ escapeReplaceNoEncode: /[<>"']|&(?!(#\d{1,7}|#[Xx][a-fA-F0-9]{1,6}|\w+);)/g,
54
+ caret: /(^|[^\[])\^/g,
55
+ percentDecode: /%25/g,
56
+ findPipe: /\|/g,
57
+ splitPipe: / \|/,
58
+ slashPipe: /\\\|/g,
59
+ carriageReturn: /\r\n|\r/g,
60
+ notSpaceStart: /^\S*/,
61
+ endingNewline: /\n$/,
62
+ listItemRegex: (bull: string) => new RegExp(`^( {0,3}${bull})((?:[\t ][^\\n]*)?(?:\\n|$))`),
63
+ nextBulletRegex: cachedIndentRegex(
64
+ (indent: number) =>
65
+ new RegExp(`^ {0,${indent}}(?:[*+-]|\\d{1,9}[.)])((?:[ \t][^\\n]*)?(?:\\n|$))`),
66
+ ),
67
+ hrRegex: cachedIndentRegex(
68
+ (indent: number) =>
69
+ new RegExp(`^ {0,${indent}}((?:- *){3,}|(?:_ *){3,}|(?:\\* *){3,})(?:\\n+|$)`),
70
+ ),
71
+ fencesBeginRegex: cachedIndentRegex(
72
+ (indent: number) => new RegExp(`^ {0,${indent}}(?:\`\`\`|~~~)`),
73
+ ),
74
+ headingBeginRegex: cachedIndentRegex((indent: number) => new RegExp(`^ {0,${indent}}#`)),
75
+ htmlBeginRegex: cachedIndentRegex(
76
+ (indent: number) => new RegExp(`^ {0,${indent}}<(?:[a-z].*>|!--)`, "i"),
77
+ ),
78
+ blockquoteBeginRegex: cachedIndentRegex((indent: number) => new RegExp(`^ {0,${indent}}>`)),
79
+ };
@@ -1,4 +1,4 @@
1
- const caret = /(^|[^\[])\^/g;
1
+ import { other } from "./other.ts";
2
2
 
3
3
  export function edit(rule: RegExp | string, opt?: string) {
4
4
  let source = typeof rule === "string" ? rule : rule.source;
@@ -6,7 +6,7 @@ export function edit(rule: RegExp | string, opt?: string) {
6
6
  const obj = {
7
7
  replace: (name: string | RegExp, val: string | RegExp) => {
8
8
  let valSource = typeof val === "string" ? val : val.source;
9
- valSource = valSource.replace(caret, "$1");
9
+ valSource = valSource.replace(other.caret, "$1");
10
10
  source = source.replace(name, valSource);
11
11
  return obj;
12
12
  },
@@ -0,0 +1,57 @@
1
+ import { type Tokens } from "./types.ts";
2
+
3
+ /**
4
+ * TextRenderer
5
+ * returns only the textual part of the token
6
+ */
7
+ export class TextRenderer {
8
+ // no need for block level renderers
9
+ strong({ text }: Tokens["Strong"]): string {
10
+ return text;
11
+ }
12
+
13
+ em({ text }: Tokens["Em"]): string {
14
+ return text;
15
+ }
16
+
17
+ codespan({ text }: Tokens["Codespan"]): string {
18
+ return text;
19
+ }
20
+
21
+ del({ text }: Tokens["Del"]): string {
22
+ return text;
23
+ }
24
+
25
+ html({ text }: Tokens["HTML"] | Tokens["Tag"]): string {
26
+ return text;
27
+ }
28
+
29
+ text({ text }: Tokens["Text"] | Tokens["Escape"] | Tokens["Tag"]): string {
30
+ return text;
31
+ }
32
+
33
+ link({ text }: Tokens["Link"]): string {
34
+ return text;
35
+ }
36
+
37
+ image({ text }: Tokens["Image"]): string {
38
+ return text;
39
+ }
40
+
41
+ br(_token: Tokens["Br"]): string {
42
+ return "";
43
+ }
44
+
45
+ checkbox({ raw }: Tokens["Checkbox"]): string {
46
+ return raw;
47
+ }
48
+
49
+ // Pantsdown extras (no upstream counterpart)
50
+ footnoteRef({ label }: Tokens["FootnoteRef"]): string {
51
+ return label;
52
+ }
53
+
54
+ latexInline({ text }: Tokens["LatexInline"]): string {
55
+ return text;
56
+ }
57
+ }