ata-validator 1.38.0 → 1.39.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -9
- package/index.d.ts +2 -1
- package/index.js +12 -28
- package/lib/aot-impl.js +40 -15
- package/lib/buffer-gate.js +6 -7
- package/lib/enrich-error.js +4 -2
- package/lib/js-compiler.js +329 -46
- package/lib/rejections.js +84 -39
- package/lib/render-pretty.js +15 -11
- package/lib/safe-regex.js +424 -413
- package/lib/schema-order.js +18 -2
- package/lib/validator-core.js +99 -13
- package/lib/version.js +1 -1
- package/package.json +11 -11
- package/lib/safe-regex-source.js +0 -8
package/lib/safe-regex.js
CHANGED
|
@@ -14,475 +14,486 @@
|
|
|
14
14
|
// supported by linear engines; compileSafe throws on them so the caller can
|
|
15
15
|
// decide (ata's codegen rejects such schemas rather than risk a hang).
|
|
16
16
|
|
|
17
|
-
//
|
|
18
|
-
//
|
|
19
|
-
//
|
|
20
|
-
//
|
|
21
|
-
//
|
|
22
|
-
|
|
23
|
-
//
|
|
24
|
-
//
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
const
|
|
32
|
-
const
|
|
33
|
-
const
|
|
34
|
-
|
|
35
|
-
function
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
17
|
+
// The engine is one function so it has one source. This module calls it, and
|
|
18
|
+
// a standalone module embeds its text (lib/aot-impl.js), the way
|
|
19
|
+
// lib/branch-collapse.js shares __ataCollapse. The embed used to come from a
|
|
20
|
+
// generated copy of this file as a string, which put the engine in every
|
|
21
|
+
// install twice.
|
|
22
|
+
function __ataSafeRegex () {
|
|
23
|
+
// ECMA-262 WhiteSpace and LineTerminator, what `\s` matches: tab, line feed,
|
|
24
|
+
// vertical tab, form feed, carriage return, space, no-break space, the
|
|
25
|
+
// Unicode space separators, the line and paragraph separators, and the byte
|
|
26
|
+
// order mark. The set used to stop at U+00A0, so `^\S+$` accepted a string
|
|
27
|
+
// holding U+2028 or U+3000.
|
|
28
|
+
const WS = [[9, 13], [32, 32], [160, 160], [0x1680, 0x1680], [0x2000, 0x200a], [0x2028, 0x2029], [0x202f, 0x202f], [0x205f, 0x205f], [0x3000, 0x3000], [0xfeff, 0xfeff]]
|
|
29
|
+
// `.` matches anything but a line terminator: LF, CR and U+2028/U+2029. It
|
|
30
|
+
// excluded only LF, so `^.$` accepted "\r".
|
|
31
|
+
const isLineTerminator = (c) => c === 10 || c === 13 || c === 0x2028 || c === 0x2029
|
|
32
|
+
const DIGIT = [[48, 57]]
|
|
33
|
+
const WORD = [[48, 57], [65, 90], [97, 122], [95, 95]]
|
|
34
|
+
|
|
35
|
+
function parse (src) {
|
|
36
|
+
let i = 0
|
|
37
|
+
const len = src.length
|
|
38
|
+
const peek = () => src[i]
|
|
39
|
+
const eof = () => i >= len
|
|
40
|
+
|
|
41
|
+
function parseAlt () {
|
|
42
|
+
const opts = [parseConcat()]
|
|
43
|
+
while (!eof() && peek() === '|') { i++; opts.push(parseConcat()) }
|
|
44
|
+
return opts.length === 1 ? opts[0] : { t: 'alt', opts }
|
|
45
|
+
}
|
|
40
46
|
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
+
function parseConcat () {
|
|
48
|
+
const parts = []
|
|
49
|
+
while (!eof() && peek() !== '|' && peek() !== ')') parts.push(parseRepeat())
|
|
50
|
+
if (parts.length === 0) return { t: 'empty' }
|
|
51
|
+
return parts.length === 1 ? parts[0] : { t: 'concat', parts }
|
|
52
|
+
}
|
|
47
53
|
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
54
|
+
function parseRepeat () {
|
|
55
|
+
let node = parseAtom()
|
|
56
|
+
while (!eof()) {
|
|
57
|
+
const ch = peek()
|
|
58
|
+
if (ch === '*') { i++; node = { t: 'star', child: node } }
|
|
59
|
+
else if (ch === '+') { i++; node = { t: 'plus', child: node } }
|
|
60
|
+
else if (ch === '?') { i++; node = { t: 'quest', child: node } }
|
|
61
|
+
else if (ch === '{') {
|
|
62
|
+
const saved = i
|
|
63
|
+
const q = tryQuantifier()
|
|
64
|
+
if (!q) { i = saved; break }
|
|
65
|
+
node = { t: 'repeat', child: node, min: q.min, max: q.max }
|
|
66
|
+
} else break
|
|
67
|
+
// a trailing ? makes the quantifier lazy; same language for a boolean test
|
|
68
|
+
if (!eof() && peek() === '?') i++
|
|
69
|
+
}
|
|
70
|
+
return node
|
|
63
71
|
}
|
|
64
|
-
return node
|
|
65
|
-
}
|
|
66
72
|
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
i++
|
|
70
|
-
let min = ''
|
|
71
|
-
while (!eof() && /[0-9]/.test(peek())) { min += peek(); i++ }
|
|
72
|
-
if (min === '') return null
|
|
73
|
-
let max
|
|
74
|
-
if (peek() === '}') { i++; return { min: +min, max: +min } }
|
|
75
|
-
if (peek() === ',') {
|
|
76
|
-
i++
|
|
77
|
-
let m = ''
|
|
78
|
-
while (!eof() && /[0-9]/.test(peek())) { m += peek(); i++ }
|
|
79
|
-
if (peek() !== '}') return null
|
|
73
|
+
function tryQuantifier () {
|
|
74
|
+
// assumes current char is '{'
|
|
80
75
|
i++
|
|
81
|
-
|
|
82
|
-
|
|
76
|
+
let min = ''
|
|
77
|
+
while (!eof() && /[0-9]/.test(peek())) { min += peek(); i++ }
|
|
78
|
+
if (min === '') return null
|
|
79
|
+
let max
|
|
80
|
+
if (peek() === '}') { i++; return { min: +min, max: +min } }
|
|
81
|
+
if (peek() === ',') {
|
|
82
|
+
i++
|
|
83
|
+
let m = ''
|
|
84
|
+
while (!eof() && /[0-9]/.test(peek())) { m += peek(); i++ }
|
|
85
|
+
if (peek() !== '}') return null
|
|
86
|
+
i++
|
|
87
|
+
max = m === '' ? Infinity : +m
|
|
88
|
+
return { min: +min, max }
|
|
89
|
+
}
|
|
90
|
+
return null
|
|
83
91
|
}
|
|
84
|
-
return null
|
|
85
|
-
}
|
|
86
92
|
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
93
|
+
function parseAtom () {
|
|
94
|
+
const ch = peek()
|
|
95
|
+
if (ch === '(') {
|
|
96
|
+
i++
|
|
97
|
+
if (src[i] === '?') {
|
|
98
|
+
if (src[i + 1] === ':') { i += 2 }
|
|
99
|
+
else throw new Error('unsupported group (lookaround/named) in pattern')
|
|
100
|
+
}
|
|
101
|
+
const child = parseAlt()
|
|
102
|
+
if (peek() !== ')') throw new Error('unbalanced ( in pattern')
|
|
103
|
+
i++
|
|
104
|
+
return { t: 'group', child }
|
|
94
105
|
}
|
|
95
|
-
|
|
96
|
-
if (
|
|
106
|
+
if (ch === '[') return parseClass()
|
|
107
|
+
if (ch === '.') { i++; return { t: 'any' } }
|
|
108
|
+
if (ch === '^') { i++; return { t: 'bol' } }
|
|
109
|
+
if (ch === '$') { i++; return { t: 'eol' } }
|
|
110
|
+
if (ch === '\\') return parseEscape(false)
|
|
111
|
+
if (ch === ')' || ch === '|') return { t: 'empty' }
|
|
97
112
|
i++
|
|
98
|
-
return { t: '
|
|
113
|
+
return { t: 'char', c: ch.charCodeAt(0) }
|
|
99
114
|
}
|
|
100
|
-
if (ch === '[') return parseClass()
|
|
101
|
-
if (ch === '.') { i++; return { t: 'any' } }
|
|
102
|
-
if (ch === '^') { i++; return { t: 'bol' } }
|
|
103
|
-
if (ch === '$') { i++; return { t: 'eol' } }
|
|
104
|
-
if (ch === '\\') return parseEscape(false)
|
|
105
|
-
if (ch === ')' || ch === '|') return { t: 'empty' }
|
|
106
|
-
i++
|
|
107
|
-
return { t: 'char', c: ch.charCodeAt(0) }
|
|
108
|
-
}
|
|
109
115
|
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
116
|
+
function parseClass () {
|
|
117
|
+
i++ // [
|
|
118
|
+
let neg = false
|
|
119
|
+
if (peek() === '^') { neg = true; i++ }
|
|
120
|
+
const ranges = []
|
|
121
|
+
while (!eof() && peek() !== ']') {
|
|
122
|
+
let lo
|
|
123
|
+
if (peek() === '\\') {
|
|
124
|
+
const esc = parseEscape(true)
|
|
125
|
+
if (esc.t === 'classpart') { for (const r of esc.ranges) ranges.push(r); continue }
|
|
126
|
+
lo = esc.c
|
|
127
|
+
} else { lo = peek().charCodeAt(0); i++ }
|
|
128
|
+
if (peek() === '-' && src[i + 1] !== ']' && i + 1 < len) {
|
|
129
|
+
i++ // -
|
|
130
|
+
let hi
|
|
131
|
+
if (peek() === '\\') { const e = parseEscape(true); hi = e.c } else { hi = peek().charCodeAt(0); i++ }
|
|
132
|
+
ranges.push([lo, hi])
|
|
133
|
+
} else {
|
|
134
|
+
ranges.push([lo, lo])
|
|
135
|
+
}
|
|
129
136
|
}
|
|
137
|
+
if (peek() !== ']') throw new Error('unbalanced [ in pattern')
|
|
138
|
+
i++
|
|
139
|
+
return { t: 'class', neg, ranges }
|
|
130
140
|
}
|
|
131
|
-
if (peek() !== ']') throw new Error('unbalanced [ in pattern')
|
|
132
|
-
i++
|
|
133
|
-
return { t: 'class', neg, ranges }
|
|
134
|
-
}
|
|
135
141
|
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
142
|
+
function parseEscape (inClass) {
|
|
143
|
+
i++ // backslash
|
|
144
|
+
if (eof()) throw new Error('trailing backslash in pattern')
|
|
145
|
+
const ch = peek(); i++
|
|
146
|
+
switch (ch) {
|
|
147
|
+
case 'd': return inClass ? { t: 'classpart', ranges: DIGIT } : { t: 'class', neg: false, ranges: DIGIT }
|
|
148
|
+
case 'w': return inClass ? { t: 'classpart', ranges: WORD } : { t: 'class', neg: false, ranges: WORD }
|
|
149
|
+
case 's': return inClass ? { t: 'classpart', ranges: WS } : { t: 'class', neg: false, ranges: WS }
|
|
150
|
+
case 'D': if (inClass) throw new Error('\\D inside a class is not supported'); return { t: 'class', neg: true, ranges: DIGIT }
|
|
151
|
+
case 'W': if (inClass) throw new Error('\\W inside a class is not supported'); return { t: 'class', neg: true, ranges: WORD }
|
|
152
|
+
case 'S': if (inClass) throw new Error('\\S inside a class is not supported'); return { t: 'class', neg: true, ranges: WS }
|
|
153
|
+
case 'n': return { t: 'char', c: 10 }
|
|
154
|
+
case 'r': return { t: 'char', c: 13 }
|
|
155
|
+
case 't': return { t: 'char', c: 9 }
|
|
156
|
+
case 'f': return { t: 'char', c: 12 }
|
|
157
|
+
case 'v': return { t: 'char', c: 11 }
|
|
158
|
+
case '0': return { t: 'char', c: 0 }
|
|
159
|
+
case 'x': { const h = src.slice(i, i + 2); i += 2; return { t: 'char', c: parseInt(h, 16) } }
|
|
160
|
+
case 'u': { const h = src.slice(i, i + 4); i += 4; return { t: 'char', c: parseInt(h, 16) } }
|
|
161
|
+
case 'b': if (inClass) return { t: 'char', c: 8 }; throw new Error('\\b word boundary is not supported')
|
|
162
|
+
default:
|
|
163
|
+
if (/[1-9]/.test(ch)) throw new Error('backreferences are not supported in pattern')
|
|
164
|
+
return { t: 'char', c: ch.charCodeAt(0) }
|
|
165
|
+
}
|
|
159
166
|
}
|
|
160
|
-
}
|
|
161
167
|
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
}
|
|
168
|
+
const ast = parseAlt()
|
|
169
|
+
if (!eof()) throw new Error('unexpected "' + peek() + '" in pattern')
|
|
170
|
+
return ast
|
|
171
|
+
}
|
|
166
172
|
|
|
167
|
-
function compileProg (ast) {
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
173
|
+
function compileProg (ast) {
|
|
174
|
+
const prog = []
|
|
175
|
+
const emit = (op, extra) => { const idx = prog.length; prog.push(Object.assign({ op }, extra)); return idx }
|
|
176
|
+
|
|
177
|
+
function rec (n) {
|
|
178
|
+
switch (n.t) {
|
|
179
|
+
case 'empty': break
|
|
180
|
+
case 'char': emit('char', { c: n.c }); break
|
|
181
|
+
case 'any': emit('any'); break
|
|
182
|
+
case 'class': emit('class', { neg: n.neg, ranges: n.ranges }); break
|
|
183
|
+
case 'bol': emit('bol'); break
|
|
184
|
+
case 'eol': emit('eol'); break
|
|
185
|
+
case 'group': rec(n.child); break
|
|
186
|
+
case 'concat': for (const p of n.parts) rec(p); break
|
|
187
|
+
case 'alt': {
|
|
188
|
+
const jmps = []
|
|
189
|
+
for (let k = 0; k < n.opts.length; k++) {
|
|
190
|
+
if (k < n.opts.length - 1) {
|
|
191
|
+
const sp = emit('split', { x: 0, y: 0 })
|
|
192
|
+
prog[sp].x = prog.length
|
|
193
|
+
rec(n.opts[k])
|
|
194
|
+
jmps.push(emit('jmp', { x: 0 }))
|
|
195
|
+
prog[sp].y = prog.length
|
|
196
|
+
} else {
|
|
197
|
+
rec(n.opts[k])
|
|
198
|
+
}
|
|
192
199
|
}
|
|
200
|
+
for (const j of jmps) prog[j].x = prog.length
|
|
201
|
+
break
|
|
193
202
|
}
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
for (let k = 0; k < n.max - n.min; k++) rec({ t: 'quest', child: n.child })
|
|
203
|
+
case 'star': {
|
|
204
|
+
const sp = emit('split', { x: 0, y: 0 })
|
|
205
|
+
prog[sp].x = prog.length
|
|
206
|
+
rec(n.child)
|
|
207
|
+
emit('jmp', { x: sp })
|
|
208
|
+
prog[sp].y = prog.length
|
|
209
|
+
break
|
|
210
|
+
}
|
|
211
|
+
case 'plus': {
|
|
212
|
+
const start = prog.length
|
|
213
|
+
rec(n.child)
|
|
214
|
+
const sp = emit('split', { x: start, y: 0 })
|
|
215
|
+
prog[sp].y = prog.length
|
|
216
|
+
break
|
|
217
|
+
}
|
|
218
|
+
case 'quest': {
|
|
219
|
+
const sp = emit('split', { x: 0, y: 0 })
|
|
220
|
+
prog[sp].x = prog.length
|
|
221
|
+
rec(n.child)
|
|
222
|
+
prog[sp].y = prog.length
|
|
223
|
+
break
|
|
224
|
+
}
|
|
225
|
+
case 'repeat': {
|
|
226
|
+
for (let k = 0; k < n.min; k++) rec(n.child)
|
|
227
|
+
if (n.max === Infinity) {
|
|
228
|
+
if (n.min === 0) rec({ t: 'star', child: n.child })
|
|
229
|
+
else rec({ t: 'star', child: n.child })
|
|
230
|
+
} else {
|
|
231
|
+
for (let k = 0; k < n.max - n.min; k++) rec({ t: 'quest', child: n.child })
|
|
232
|
+
}
|
|
233
|
+
break
|
|
226
234
|
}
|
|
227
|
-
break
|
|
228
235
|
}
|
|
229
236
|
}
|
|
230
|
-
}
|
|
231
|
-
|
|
232
|
-
rec(ast)
|
|
233
|
-
emit('match')
|
|
234
|
-
return prog
|
|
235
|
-
}
|
|
236
237
|
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
const OP_ANY = 1
|
|
241
|
-
const OP_CLASS = 2
|
|
242
|
-
const OP_SPLIT = 3
|
|
243
|
-
const OP_JMP = 4
|
|
244
|
-
const OP_BOL = 5
|
|
245
|
-
const OP_EOL = 6
|
|
246
|
-
const OP_MATCH = 7
|
|
247
|
-
|
|
248
|
-
function classMatcher (instr) {
|
|
249
|
-
// ASCII is answered from a bitmap; anything above 0x7f walks the ranges.
|
|
250
|
-
const bits = new Uint8Array(128)
|
|
251
|
-
const r = instr.ranges
|
|
252
|
-
for (let k = 0; k < r.length; k++) {
|
|
253
|
-
const hi = Math.min(r[k][1], 127)
|
|
254
|
-
for (let c = r[k][0]; c <= hi; c++) bits[c] = 1
|
|
238
|
+
rec(ast)
|
|
239
|
+
emit('match')
|
|
240
|
+
return prog
|
|
255
241
|
}
|
|
256
|
-
return { bits, ranges: r, neg: instr.neg }
|
|
257
|
-
}
|
|
258
242
|
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
243
|
+
// Numeric opcodes for the runner. The program is compiled once into flat
|
|
244
|
+
// typed arrays so the inner loop does no property lookups or string compares.
|
|
245
|
+
const OP_CHAR = 0
|
|
246
|
+
const OP_ANY = 1
|
|
247
|
+
const OP_CLASS = 2
|
|
248
|
+
const OP_SPLIT = 3
|
|
249
|
+
const OP_JMP = 4
|
|
250
|
+
const OP_BOL = 5
|
|
251
|
+
const OP_EOL = 6
|
|
252
|
+
const OP_MATCH = 7
|
|
253
|
+
|
|
254
|
+
function classMatcher (instr) {
|
|
255
|
+
// ASCII is answered from a bitmap; anything above 0x7f walks the ranges.
|
|
256
|
+
const bits = new Uint8Array(128)
|
|
257
|
+
const r = instr.ranges
|
|
258
|
+
for (let k = 0; k < r.length; k++) {
|
|
259
|
+
const hi = Math.min(r[k][1], 127)
|
|
260
|
+
for (let c = r[k][0]; c <= hi; c++) bits[c] = 1
|
|
261
|
+
}
|
|
262
|
+
return { bits, ranges: r, neg: instr.neg }
|
|
267
263
|
}
|
|
268
|
-
return cls.neg ? !inside : inside
|
|
269
|
-
}
|
|
270
264
|
|
|
271
|
-
function
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
const I = prog[i]
|
|
280
|
-
switch (I.op) {
|
|
281
|
-
case 'char': ops[i] = OP_CHAR; cs[i] = I.c; break
|
|
282
|
-
case 'any': ops[i] = OP_ANY; break
|
|
283
|
-
case 'class': ops[i] = OP_CLASS; classes[i] = classMatcher(I); break
|
|
284
|
-
case 'split': ops[i] = OP_SPLIT; xs[i] = I.x; ys[i] = I.y; break
|
|
285
|
-
case 'jmp': ops[i] = OP_JMP; xs[i] = I.x; break
|
|
286
|
-
case 'bol': ops[i] = OP_BOL; break
|
|
287
|
-
case 'eol': ops[i] = OP_EOL; break
|
|
288
|
-
case 'match': ops[i] = OP_MATCH; break
|
|
265
|
+
function matchClass (cls, c) {
|
|
266
|
+
let inside
|
|
267
|
+
if (c < 128) {
|
|
268
|
+
inside = cls.bits[c] === 1
|
|
269
|
+
} else {
|
|
270
|
+
inside = false
|
|
271
|
+
const r = cls.ranges
|
|
272
|
+
for (let k = 0; k < r.length; k++) { if (c >= r[k][0] && c <= r[k][1]) { inside = true; break } }
|
|
289
273
|
}
|
|
274
|
+
return cls.neg ? !inside : inside
|
|
290
275
|
}
|
|
291
276
|
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
lastGen[pc] = gen
|
|
311
|
-
list[len0] = pc
|
|
312
|
-
return len0 + 1
|
|
313
|
-
}
|
|
314
|
-
let sp = 0
|
|
315
|
-
stack[sp++] = pc
|
|
316
|
-
let count = len0
|
|
317
|
-
while (sp > 0) {
|
|
318
|
-
const p = stack[--sp]
|
|
319
|
-
if (lastGen[p] === gen) continue
|
|
320
|
-
lastGen[p] = gen
|
|
321
|
-
switch (ops[p]) {
|
|
322
|
-
case OP_JMP: stack[sp++] = xs[p]; break
|
|
323
|
-
case OP_SPLIT: stack[sp++] = ys[p]; stack[sp++] = xs[p]; break
|
|
324
|
-
case OP_BOL: if (pos === 0) stack[sp++] = p + 1; break
|
|
325
|
-
case OP_EOL: if (pos === len) stack[sp++] = p + 1; break
|
|
326
|
-
default: list[count++] = p
|
|
277
|
+
function makeRunner (prog) {
|
|
278
|
+
const n = prog.length
|
|
279
|
+
const ops = new Uint8Array(n)
|
|
280
|
+
const xs = new Int32Array(n)
|
|
281
|
+
const ys = new Int32Array(n)
|
|
282
|
+
const cs = new Int32Array(n)
|
|
283
|
+
const classes = new Array(n)
|
|
284
|
+
for (let i = 0; i < n; i++) {
|
|
285
|
+
const I = prog[i]
|
|
286
|
+
switch (I.op) {
|
|
287
|
+
case 'char': ops[i] = OP_CHAR; cs[i] = I.c; break
|
|
288
|
+
case 'any': ops[i] = OP_ANY; break
|
|
289
|
+
case 'class': ops[i] = OP_CLASS; classes[i] = classMatcher(I); break
|
|
290
|
+
case 'split': ops[i] = OP_SPLIT; xs[i] = I.x; ys[i] = I.y; break
|
|
291
|
+
case 'jmp': ops[i] = OP_JMP; xs[i] = I.x; break
|
|
292
|
+
case 'bol': ops[i] = OP_BOL; break
|
|
293
|
+
case 'eol': ops[i] = OP_EOL; break
|
|
294
|
+
case 'match': ops[i] = OP_MATCH; break
|
|
327
295
|
}
|
|
328
296
|
}
|
|
329
|
-
return count
|
|
330
|
-
}
|
|
331
297
|
|
|
332
|
-
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
|
|
298
|
+
const lastGen = new Int32Array(n).fill(-1)
|
|
299
|
+
let gen = 0
|
|
300
|
+
// Each unvisited instruction is popped once and pushes at most two, so the
|
|
301
|
+
// stack never holds more than 2n + 1 entries.
|
|
302
|
+
const stack = new Int32Array(2 * n + 2)
|
|
303
|
+
// Thread lists hold at most one entry per instruction per step.
|
|
304
|
+
let clist = new Int32Array(n)
|
|
305
|
+
let nlist = new Int32Array(n)
|
|
306
|
+
let clen = 0
|
|
307
|
+
let nlen = 0
|
|
308
|
+
|
|
309
|
+
// Follows epsilon edges from `pc` and records every consuming instruction
|
|
310
|
+
// (or match) reached in `list`. `lastGen` dedupes per step.
|
|
311
|
+
function addThread (list, len0, pc, pos, len) {
|
|
312
|
+
// Most transitions land directly on a consuming instruction; skip the
|
|
313
|
+
// stack walk for those.
|
|
314
|
+
if (ops[pc] <= OP_CLASS || ops[pc] === OP_MATCH) {
|
|
315
|
+
if (lastGen[pc] === gen) return len0
|
|
316
|
+
lastGen[pc] = gen
|
|
317
|
+
list[len0] = pc
|
|
318
|
+
return len0 + 1
|
|
319
|
+
}
|
|
320
|
+
let sp = 0
|
|
321
|
+
stack[sp++] = pc
|
|
322
|
+
let count = len0
|
|
323
|
+
while (sp > 0) {
|
|
324
|
+
const p = stack[--sp]
|
|
325
|
+
if (lastGen[p] === gen) continue
|
|
326
|
+
lastGen[p] = gen
|
|
327
|
+
switch (ops[p]) {
|
|
328
|
+
case OP_JMP: stack[sp++] = xs[p]; break
|
|
329
|
+
case OP_SPLIT: stack[sp++] = ys[p]; stack[sp++] = xs[p]; break
|
|
330
|
+
case OP_BOL: if (pos === 0) stack[sp++] = p + 1; break
|
|
331
|
+
case OP_EOL: if (pos === len) stack[sp++] = p + 1; break
|
|
332
|
+
default: list[count++] = p
|
|
333
|
+
}
|
|
334
|
+
}
|
|
335
|
+
return count
|
|
336
|
+
}
|
|
339
337
|
|
|
340
|
-
|
|
341
|
-
|
|
338
|
+
// A pattern is anchored when starting it anywhere but position 0 yields no
|
|
339
|
+
// thread, which is the case for `^...` and its alternations. The probe sits
|
|
340
|
+
// at position 1 of a length-1 string so that only `^` can fail. For anchored
|
|
341
|
+
// patterns the per-position restart below is skipped, and an empty thread
|
|
342
|
+
// list means the match has already failed.
|
|
342
343
|
gen++
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
344
|
+
const anchored = addThread(nlist, 0, 0, 1, 1) === 0
|
|
345
|
+
|
|
346
|
+
function testNFA (s) {
|
|
347
|
+
const len = s.length
|
|
346
348
|
gen++
|
|
347
|
-
|
|
348
|
-
for (let
|
|
349
|
-
const
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
349
|
+
clen = addThread(clist, 0, 0, 0, len)
|
|
350
|
+
for (let pos = 0; pos <= len; pos++) {
|
|
351
|
+
const c = pos < len ? s.charCodeAt(pos) : -1
|
|
352
|
+
gen++
|
|
353
|
+
nlen = 0
|
|
354
|
+
for (let k = 0; k < clen; k++) {
|
|
355
|
+
const pc = clist[k]
|
|
356
|
+
switch (ops[pc]) {
|
|
357
|
+
case OP_MATCH: return true
|
|
358
|
+
case OP_CHAR: if (c === cs[pc]) nlen = addThread(nlist, nlen, pc + 1, pos + 1, len); break
|
|
359
|
+
case OP_ANY: if (c !== -1 && !isLineTerminator(c)) nlen = addThread(nlist, nlen, pc + 1, pos + 1, len); break
|
|
360
|
+
case OP_CLASS: if (c !== -1 && matchClass(classes[pc], c)) nlen = addThread(nlist, nlen, pc + 1, pos + 1, len); break
|
|
361
|
+
}
|
|
355
362
|
}
|
|
363
|
+
if (pos < len) {
|
|
364
|
+
if (!anchored) nlen = addThread(nlist, nlen, 0, pos + 1, len)
|
|
365
|
+
else if (nlen === 0) return false
|
|
366
|
+
}
|
|
367
|
+
const tmp = clist; clist = nlist; nlist = tmp
|
|
368
|
+
clen = nlen
|
|
356
369
|
}
|
|
357
|
-
|
|
358
|
-
if (!anchored) nlen = addThread(nlist, nlen, 0, pos + 1, len)
|
|
359
|
-
else if (nlen === 0) return false
|
|
360
|
-
}
|
|
361
|
-
const tmp = clist; clist = nlist; nlist = tmp
|
|
362
|
-
clen = nlen
|
|
370
|
+
return false
|
|
363
371
|
}
|
|
364
|
-
return false
|
|
365
|
-
}
|
|
366
372
|
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
|
|
386
|
-
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
|
|
391
|
-
|
|
392
|
-
|
|
373
|
+
// Lazy DFA on top of the NFA. A DFA state is the set of consuming
|
|
374
|
+
// instructions live at a position; transitions are computed on first use
|
|
375
|
+
// and cached per ASCII character. `^` and `$` depend on position, so the
|
|
376
|
+
// closure is taken with flags for "at start" and "at end", which gives two
|
|
377
|
+
// start states and two transition tables per state. The state count is
|
|
378
|
+
// capped; past the cap the matcher falls back to the NFA walk above, so the
|
|
379
|
+
// time bound stays linear either way.
|
|
380
|
+
const MAX_STATES = 256
|
|
381
|
+
const states = []
|
|
382
|
+
const stateIds = new Map()
|
|
383
|
+
let overflow = false
|
|
384
|
+
|
|
385
|
+
function closure (list, count, pc, atStart, atEnd) {
|
|
386
|
+
// Same walk as addThread, with the position replaced by the two flags.
|
|
387
|
+
let sp = 0
|
|
388
|
+
stack[sp++] = pc
|
|
389
|
+
while (sp > 0) {
|
|
390
|
+
const p = stack[--sp]
|
|
391
|
+
if (lastGen[p] === gen) continue
|
|
392
|
+
lastGen[p] = gen
|
|
393
|
+
switch (ops[p]) {
|
|
394
|
+
case OP_JMP: stack[sp++] = xs[p]; break
|
|
395
|
+
case OP_SPLIT: stack[sp++] = ys[p]; stack[sp++] = xs[p]; break
|
|
396
|
+
case OP_BOL: if (atStart) stack[sp++] = p + 1; break
|
|
397
|
+
case OP_EOL: if (atEnd) stack[sp++] = p + 1; break
|
|
398
|
+
default: list[count++] = p
|
|
399
|
+
}
|
|
393
400
|
}
|
|
401
|
+
return count
|
|
394
402
|
}
|
|
395
|
-
return count
|
|
396
|
-
}
|
|
397
403
|
|
|
398
|
-
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
|
|
404
|
-
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
|
|
404
|
+
function internState (list, count) {
|
|
405
|
+
const pcs = Array.from(list.subarray(0, count)).sort((a, b) => a - b)
|
|
406
|
+
const key = pcs.join(',')
|
|
407
|
+
let id = stateIds.get(key)
|
|
408
|
+
if (id !== undefined) return id
|
|
409
|
+
if (states.length >= MAX_STATES) { overflow = true; return -1 }
|
|
410
|
+
id = states.length
|
|
411
|
+
let isMatch = false
|
|
412
|
+
for (let k = 0; k < pcs.length; k++) if (ops[pcs[k]] === OP_MATCH) { isMatch = true; break }
|
|
413
|
+
states.push({ pcs: Int32Array.from(pcs), isMatch, next: new Int32Array(128).fill(-2), nextEnd: new Int32Array(128).fill(-2) })
|
|
414
|
+
stateIds.set(key, id)
|
|
415
|
+
return id
|
|
416
|
+
}
|
|
411
417
|
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
|
|
421
|
-
|
|
418
|
+
function step (state, c, atEnd) {
|
|
419
|
+
gen++
|
|
420
|
+
let count = 0
|
|
421
|
+
const pcs = state.pcs
|
|
422
|
+
for (let k = 0; k < pcs.length; k++) {
|
|
423
|
+
const pc = pcs[k]
|
|
424
|
+
switch (ops[pc]) {
|
|
425
|
+
case OP_CHAR: if (c === cs[pc]) count = closure(nlist, count, pc + 1, false, atEnd); break
|
|
426
|
+
case OP_ANY: if (!isLineTerminator(c)) count = closure(nlist, count, pc + 1, false, atEnd); break
|
|
427
|
+
case OP_CLASS: if (matchClass(classes[pc], c)) count = closure(nlist, count, pc + 1, false, atEnd); break
|
|
428
|
+
}
|
|
422
429
|
}
|
|
430
|
+
if (!anchored) count = closure(nlist, count, 0, false, atEnd)
|
|
431
|
+
return internState(nlist, count)
|
|
423
432
|
}
|
|
424
|
-
if (!anchored) count = closure(nlist, count, 0, false, atEnd)
|
|
425
|
-
return internState(nlist, count)
|
|
426
|
-
}
|
|
427
433
|
|
|
428
|
-
|
|
429
|
-
|
|
434
|
+
let startEmpty = -2
|
|
435
|
+
let startNonEmpty = -2
|
|
430
436
|
|
|
431
|
-
|
|
432
|
-
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
}
|
|
436
|
-
|
|
437
|
-
function testDFA (s) {
|
|
438
|
-
const len = s.length
|
|
439
|
-
let id
|
|
440
|
-
if (len === 0) {
|
|
441
|
-
if (startEmpty === -2) startEmpty = startState(true)
|
|
442
|
-
id = startEmpty
|
|
443
|
-
} else {
|
|
444
|
-
if (startNonEmpty === -2) startNonEmpty = startState(false)
|
|
445
|
-
id = startNonEmpty
|
|
437
|
+
function startState (atEnd) {
|
|
438
|
+
gen++
|
|
439
|
+
const count = closure(nlist, 0, 0, true, atEnd)
|
|
440
|
+
return internState(nlist, count)
|
|
446
441
|
}
|
|
447
|
-
|
|
448
|
-
|
|
449
|
-
|
|
450
|
-
|
|
451
|
-
|
|
452
|
-
|
|
453
|
-
|
|
454
|
-
if (c < 128) {
|
|
455
|
-
const table = atEnd ? state.nextEnd : state.next
|
|
456
|
-
nid = table[c]
|
|
457
|
-
if (nid === -2) { nid = step(state, c, atEnd); table[c] = nid }
|
|
442
|
+
|
|
443
|
+
function testDFA (s) {
|
|
444
|
+
const len = s.length
|
|
445
|
+
let id
|
|
446
|
+
if (len === 0) {
|
|
447
|
+
if (startEmpty === -2) startEmpty = startState(true)
|
|
448
|
+
id = startEmpty
|
|
458
449
|
} else {
|
|
459
|
-
|
|
450
|
+
if (startNonEmpty === -2) startNonEmpty = startState(false)
|
|
451
|
+
id = startNonEmpty
|
|
460
452
|
}
|
|
461
|
-
if (
|
|
462
|
-
state = states[
|
|
463
|
-
|
|
453
|
+
if (id < 0) return testNFA(s)
|
|
454
|
+
let state = states[id]
|
|
455
|
+
for (let pos = 0; pos < len; pos++) {
|
|
456
|
+
if (state.isMatch) return true
|
|
457
|
+
const c = s.charCodeAt(pos)
|
|
458
|
+
const atEnd = pos + 1 === len
|
|
459
|
+
let nid
|
|
460
|
+
if (c < 128) {
|
|
461
|
+
const table = atEnd ? state.nextEnd : state.next
|
|
462
|
+
nid = table[c]
|
|
463
|
+
if (nid === -2) { nid = step(state, c, atEnd); table[c] = nid }
|
|
464
|
+
} else {
|
|
465
|
+
nid = step(state, c, atEnd)
|
|
466
|
+
}
|
|
467
|
+
if (nid < 0) return testNFA(s)
|
|
468
|
+
state = states[nid]
|
|
469
|
+
if (anchored && state.pcs.length === 0) return false
|
|
470
|
+
}
|
|
471
|
+
return state.isMatch
|
|
472
|
+
}
|
|
473
|
+
|
|
474
|
+
return function test (s) {
|
|
475
|
+
return overflow ? testNFA(s) : testDFA(s)
|
|
464
476
|
}
|
|
465
|
-
return state.isMatch
|
|
466
477
|
}
|
|
467
478
|
|
|
468
|
-
|
|
469
|
-
|
|
479
|
+
function compileSafe (pattern) {
|
|
480
|
+
const prog = compileProg(parse(pattern))
|
|
481
|
+
const runner = makeRunner(prog)
|
|
482
|
+
// `__ataSafe` brands the result so the standalone serializer can tell a safe
|
|
483
|
+
// matcher apart from a RegExp and emit `__ataSafeRe(source)` instead.
|
|
484
|
+
return { test: runner, source: pattern, __ataSafe: true }
|
|
470
485
|
}
|
|
471
|
-
}
|
|
472
486
|
|
|
473
|
-
|
|
474
|
-
|
|
475
|
-
|
|
476
|
-
|
|
477
|
-
|
|
478
|
-
|
|
479
|
-
}
|
|
487
|
+
// True when the linear engine can represent `src`. Used by the codegen to decide
|
|
488
|
+
// between the safe matcher and a JS RegExp fallback for patterns outside the
|
|
489
|
+
// supported (RE2) subset (backreferences, lookaround, etc.).
|
|
490
|
+
function patternIsSafe (src) {
|
|
491
|
+
try { compileSafe(src); return true } catch { return false }
|
|
492
|
+
}
|
|
480
493
|
|
|
481
|
-
|
|
482
|
-
// between the safe matcher and a JS RegExp fallback for patterns outside the
|
|
483
|
-
// supported (RE2) subset (backreferences, lookaround, etc.).
|
|
484
|
-
function patternIsSafe (src) {
|
|
485
|
-
try { compileSafe(src); return true } catch { return false }
|
|
494
|
+
return { compileSafe, patternIsSafe, parse }
|
|
486
495
|
}
|
|
487
496
|
|
|
488
|
-
module.exports =
|
|
497
|
+
module.exports = __ataSafeRegex()
|
|
498
|
+
// The engine's source, for the standalone emitter.
|
|
499
|
+
module.exports.engineSource = () => __ataSafeRegex.toString()
|