ata-validator 1.6.2 → 1.7.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +40 -0
- package/README.md +15 -5
- package/build.d.ts +12 -0
- package/index.d.ts +7 -0
- package/index.js +178 -57
- package/lib/aot-build.js +1 -0
- package/lib/aot.js +70 -20
- package/lib/buffer-gate.js +132 -0
- package/lib/draft7.js +24 -2
- package/lib/interpreter.js +644 -208
- package/lib/js-compiler.js +187 -55
- package/lib/metaschemas.js +21 -0
- package/lib/plan-compiler.js +545 -0
- package/lib/safe-regex-source.js +1 -1
- package/lib/safe-regex.js +209 -29
- package/lib/version.js +1 -1
- package/package.json +8 -8
|
@@ -0,0 +1,545 @@
|
|
|
1
|
+
'use strict';
|
|
2
|
+
|
|
3
|
+
// Compiles interpreter Plans into a tree of specialized closures. No source
|
|
4
|
+
// generation, no `new Function`: this is plain closure composition, so it
|
|
5
|
+
// runs wherever the interpreter runs, CSP included. The payoff over eval()
|
|
6
|
+
// is that every keyword branch is decided once at compile time and every
|
|
7
|
+
// child call is a direct monomorphic call, which is the same advantage the
|
|
8
|
+
// code generator has, minus the codegen.
|
|
9
|
+
//
|
|
10
|
+
// Scope, deliberately narrow:
|
|
11
|
+
// - single schema resource only (no embedded `$id`, no external documents),
|
|
12
|
+
// so the base URI and the dynamic scope are compile-time constants and
|
|
13
|
+
// `$ref` / `$dynamicRef` targets resolve once, here;
|
|
14
|
+
// - no `unevaluatedProperties` / `unevaluatedItems` anywhere (annotation
|
|
15
|
+
// flow stays eval()'s job).
|
|
16
|
+
// Anything outside that gate keeps the generic evaluator. Error output must
|
|
17
|
+
// be byte-for-byte what eval() produces; tests/test_plan_compiler.js diffs
|
|
18
|
+
// the two over the official suite.
|
|
19
|
+
//
|
|
20
|
+
// Compiled signature: fn(data, errors, instancePath, schemaPath, stack) ->
|
|
21
|
+
// boolean. `errors` may be the NOERRORS sentinel for verdict-only runs, in
|
|
22
|
+
// which case a failed step returns immediately; when collecting, every step
|
|
23
|
+
// runs, exactly like eval().
|
|
24
|
+
|
|
25
|
+
function install(deps) {
|
|
26
|
+
const { Plan, NOERRORS, err, evalLeaf, evalLeafV, deepEqual, dataBits, escapePointer,
|
|
27
|
+
T_STRING, T_ARRAY, T_OBJECT, resolveRef, splitFragment, resolveUri } = deps;
|
|
28
|
+
|
|
29
|
+
const TRUE_FN = () => true;
|
|
30
|
+
const TRUE_PAIR = { v: TRUE_FN, c: TRUE_FN };
|
|
31
|
+
const FALSE_PAIR = {
|
|
32
|
+
v: () => false,
|
|
33
|
+
c: (data, errors, instancePath, schemaPath) => {
|
|
34
|
+
if (errors !== NOERRORS) errors.push(err('false schema', 'not', instancePath, schemaPath, {}, 'boolean schema is false'));
|
|
35
|
+
return false;
|
|
36
|
+
},
|
|
37
|
+
};
|
|
38
|
+
|
|
39
|
+
// Whether the whole tree reachable from `node` fits the compiler's scope.
|
|
40
|
+
// `base` is the base URI in effect, updated at compile time exactly the way
|
|
41
|
+
// eval() updates it at run time: the schema path is static, so it can be.
|
|
42
|
+
function compilable(interp, node, base, seen) {
|
|
43
|
+
if (node === true || node === false) return true;
|
|
44
|
+
if (!(node instanceof Plan)) return true;
|
|
45
|
+
const P = node;
|
|
46
|
+
if (P.nodeBase !== null && P.nodeBase !== base) base = P.nodeBase;
|
|
47
|
+
const key = P;
|
|
48
|
+
const prev = seen.get(key);
|
|
49
|
+
if (prev !== undefined) { if (prev === base || prev === true) return true; return false; }
|
|
50
|
+
seen.set(key, base);
|
|
51
|
+
if (P.hasUnevaluated) return false;
|
|
52
|
+
if (P.ref !== null) {
|
|
53
|
+
const t = resolveRef(P.ref, base, interp.state);
|
|
54
|
+
if (t.node === undefined || !compilable(interp, interp.node(t.node), t.base, seen)) return false;
|
|
55
|
+
}
|
|
56
|
+
if (P.dynamicRef !== null) {
|
|
57
|
+
// $dynamicRef stays compiled only in a single-resource schema, where
|
|
58
|
+
// the dynamic scope is a constant. Anything larger keeps eval().
|
|
59
|
+
if (interp.state.resources.size !== 1) return false;
|
|
60
|
+
const t = dynTarget(interp, P.dynamicRef);
|
|
61
|
+
if (t === undefined || !compilable(interp, interp.node(t), base, seen)) return false;
|
|
62
|
+
}
|
|
63
|
+
const kids = [];
|
|
64
|
+
if (P.properties !== null) for (const v of P.properties.values()) kids.push(v.node);
|
|
65
|
+
if (P.patternProperties !== null) for (const e of P.patternProperties) kids.push(e.node);
|
|
66
|
+
if (P.additionalProperties !== undefined) kids.push(P.additionalProperties);
|
|
67
|
+
if (P.propertyNames !== undefined) kids.push(P.propertyNames);
|
|
68
|
+
if (P.dependentSchemas !== null) for (const [, v] of P.dependentSchemas) kids.push(v);
|
|
69
|
+
if (P.propertyDependencies !== null) for (const [, m] of P.propertyDependencies) for (const v of m.values()) kids.push(v);
|
|
70
|
+
if (P.prefixItems !== null) for (const v of P.prefixItems) kids.push(v);
|
|
71
|
+
if (P.items !== undefined) kids.push(P.items);
|
|
72
|
+
if (P.contains !== undefined) kids.push(P.contains);
|
|
73
|
+
if (P.allOf !== null) for (const v of P.allOf) kids.push(v);
|
|
74
|
+
if (P.anyOf !== null) for (const v of P.anyOf) kids.push(v);
|
|
75
|
+
if (P.oneOf !== null) for (const v of P.oneOf) kids.push(v);
|
|
76
|
+
if (P.not !== undefined) kids.push(P.not);
|
|
77
|
+
if (P.if !== undefined) kids.push(P.if);
|
|
78
|
+
if (P.then !== undefined) kids.push(P.then);
|
|
79
|
+
if (P.else !== undefined) kids.push(P.else);
|
|
80
|
+
for (const k of kids) if (!compilable(interp, k, base, seen)) return false;
|
|
81
|
+
return true;
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
// Static $dynamicRef target: with a single resource the dynamic scope is
|
|
85
|
+
// always [rootBase], so the outermost-scope search collapses to a lookup.
|
|
86
|
+
function dynTarget(interp, ref) {
|
|
87
|
+
const base = interp.state.rootBase;
|
|
88
|
+
let { node } = resolveRef(ref, base, interp.state);
|
|
89
|
+
const [, fragment] = splitFragment(resolveUri(base, ref));
|
|
90
|
+
if (fragment && !fragment.startsWith('/')) {
|
|
91
|
+
const dyn = interp.state.dynamicAnchors.get(base);
|
|
92
|
+
const bookended = node !== undefined && dyn && dyn.get(fragment) === node;
|
|
93
|
+
if (bookended || !interp.bookending) {
|
|
94
|
+
if (dyn && dyn.has(fragment)) node = dyn.get(fragment);
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
return node;
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
function compileNode(interp, node, base, memo) {
|
|
101
|
+
if (node === true) return TRUE_PAIR;
|
|
102
|
+
if (node === false) return FALSE_PAIR;
|
|
103
|
+
if (!(node instanceof Plan)) return TRUE_PAIR;
|
|
104
|
+
const P = node;
|
|
105
|
+
if (P.nodeBase !== null && P.nodeBase !== base) base = P.nodeBase;
|
|
106
|
+
// A plan reached under two bases compiles once per base; nearly every
|
|
107
|
+
// plan only ever has one.
|
|
108
|
+
let perBase = memo.get(P);
|
|
109
|
+
if (perBase === undefined) { perBase = new Map(); memo.set(P, perBase); }
|
|
110
|
+
const cached = perBase.get(base);
|
|
111
|
+
if (cached !== undefined) return cached;
|
|
112
|
+
// Cycles: register trampolines before compiling children.
|
|
113
|
+
const box = { v: null, c: null };
|
|
114
|
+
const pair = {
|
|
115
|
+
v: (d, st) => box.v(d, st),
|
|
116
|
+
c: (d, e, ip, sp, st) => box.c(d, e, ip, sp, st),
|
|
117
|
+
};
|
|
118
|
+
perBase.set(base, pair);
|
|
119
|
+
|
|
120
|
+
const steps = []; // collect variant: (data, errors, ip, sp, stack)
|
|
121
|
+
const vsteps = []; // verdict variant: (data, stack), no strings at all
|
|
122
|
+
|
|
123
|
+
// $ref / $dynamicRef, resolved at compile time. The cycle guard mirrors
|
|
124
|
+
// eval(): a (schema, data) pair already on the stack is a fixed point.
|
|
125
|
+
if (P.ref !== null) {
|
|
126
|
+
const t = resolveRef(P.ref, base, interp.state);
|
|
127
|
+
const target = compileNode(interp, interp.node(t.node), t.base, memo);
|
|
128
|
+
const schema = P.schema;
|
|
129
|
+
steps.push((data, errors, instancePath, schemaPath, stack) => {
|
|
130
|
+
for (let i = stack.length - 2; i >= 0; i -= 2) {
|
|
131
|
+
if (stack[i] === schema && stack[i + 1] === data) return true;
|
|
132
|
+
}
|
|
133
|
+
stack.push(schema, data);
|
|
134
|
+
const ok = target.c(data, errors, instancePath, schemaPath + '/$ref', stack);
|
|
135
|
+
stack.length -= 2;
|
|
136
|
+
return ok;
|
|
137
|
+
});
|
|
138
|
+
vsteps.push((data, stack) => {
|
|
139
|
+
for (let i = stack.length - 2; i >= 0; i -= 2) {
|
|
140
|
+
if (stack[i] === schema && stack[i + 1] === data) return true;
|
|
141
|
+
}
|
|
142
|
+
stack.push(schema, data);
|
|
143
|
+
const ok = target.v(data, stack);
|
|
144
|
+
stack.length -= 2;
|
|
145
|
+
return ok;
|
|
146
|
+
});
|
|
147
|
+
}
|
|
148
|
+
if (P.dynamicRef !== null) {
|
|
149
|
+
const target = compileNode(interp, interp.node(dynTarget(interp, P.dynamicRef)), base, memo);
|
|
150
|
+
const schema = P.schema;
|
|
151
|
+
steps.push((data, errors, instancePath, schemaPath, stack) => {
|
|
152
|
+
for (let i = stack.length - 2; i >= 0; i -= 2) {
|
|
153
|
+
if (stack[i] === schema && stack[i + 1] === data) return true;
|
|
154
|
+
}
|
|
155
|
+
stack.push(schema, data);
|
|
156
|
+
const ok = target.c(data, errors, instancePath, schemaPath + '/$dynamicRef', stack);
|
|
157
|
+
stack.length -= 2;
|
|
158
|
+
return ok;
|
|
159
|
+
});
|
|
160
|
+
vsteps.push((data, stack) => {
|
|
161
|
+
for (let i = stack.length - 2; i >= 0; i -= 2) {
|
|
162
|
+
if (stack[i] === schema && stack[i + 1] === data) return true;
|
|
163
|
+
}
|
|
164
|
+
stack.push(schema, data);
|
|
165
|
+
const ok = target.v(data, stack);
|
|
166
|
+
stack.length -= 2;
|
|
167
|
+
return ok;
|
|
168
|
+
});
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
// Every value-level keyword in one step, reusing the leaf evaluator so
|
|
172
|
+
// the two paths cannot drift.
|
|
173
|
+
if (P.hasType || P.enum !== null || P.hasConst || P.hasNumber || P.hasString ||
|
|
174
|
+
P.minItems !== undefined || P.maxItems !== undefined || P.uniqueItems ||
|
|
175
|
+
P.required !== null || P.minProperties !== undefined || P.maxProperties !== undefined ||
|
|
176
|
+
P.dependentRequired !== null) {
|
|
177
|
+
steps.push((data, errors, instancePath, schemaPath) => evalLeaf(P, data, errors, instancePath, schemaPath));
|
|
178
|
+
vsteps.push((data) => evalLeafV(P, data));
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
// Arrays
|
|
182
|
+
if (P.prefixItems !== null) {
|
|
183
|
+
const fns = P.prefixItems.map((v) => compileNode(interp, v, base, memo));
|
|
184
|
+
steps.push((data, errors, instancePath, schemaPath, stack) => {
|
|
185
|
+
if (dataBits(data) !== T_ARRAY) return true;
|
|
186
|
+
let ok = true;
|
|
187
|
+
const n = Math.min(fns.length, data.length);
|
|
188
|
+
for (let i = 0; i < n; i++) {
|
|
189
|
+
if (!fns[i].c(data[i], errors, instancePath + '/' + i, schemaPath + '/prefixItems/' + i, stack)) {
|
|
190
|
+
ok = false;
|
|
191
|
+
if (errors === NOERRORS) return false;
|
|
192
|
+
}
|
|
193
|
+
}
|
|
194
|
+
return ok;
|
|
195
|
+
});
|
|
196
|
+
vsteps.push((data, stack) => {
|
|
197
|
+
if (dataBits(data) !== T_ARRAY) return true;
|
|
198
|
+
const n = Math.min(fns.length, data.length);
|
|
199
|
+
for (let i = 0; i < n; i++) if (!fns[i].v(data[i], stack)) return false;
|
|
200
|
+
return true;
|
|
201
|
+
});
|
|
202
|
+
}
|
|
203
|
+
if (P.items !== undefined) {
|
|
204
|
+
const fn = compileNode(interp, P.items, base, memo);
|
|
205
|
+
const start = P.prefixItems !== null ? P.prefixItems.length : 0;
|
|
206
|
+
steps.push((data, errors, instancePath, schemaPath, stack) => {
|
|
207
|
+
if (dataBits(data) !== T_ARRAY) return true;
|
|
208
|
+
let ok = true;
|
|
209
|
+
for (let i = start; i < data.length; i++) {
|
|
210
|
+
if (!fn.c(data[i], errors, instancePath + '/' + i, schemaPath + '/items', stack)) {
|
|
211
|
+
ok = false;
|
|
212
|
+
if (errors === NOERRORS) return false;
|
|
213
|
+
}
|
|
214
|
+
}
|
|
215
|
+
return ok;
|
|
216
|
+
});
|
|
217
|
+
const fv = fn.v;
|
|
218
|
+
vsteps.push((data, stack) => {
|
|
219
|
+
if (dataBits(data) !== T_ARRAY) return true;
|
|
220
|
+
for (let i = start; i < data.length; i++) if (!fv(data[i], stack)) return false;
|
|
221
|
+
return true;
|
|
222
|
+
});
|
|
223
|
+
}
|
|
224
|
+
if (P.contains !== undefined) {
|
|
225
|
+
const fn = compileNode(interp, P.contains, base, memo);
|
|
226
|
+
const minC = P.minContains !== undefined ? P.minContains : 1;
|
|
227
|
+
const maxC = P.maxContains;
|
|
228
|
+
steps.push((data, errors, instancePath, schemaPath, stack) => {
|
|
229
|
+
if (dataBits(data) !== T_ARRAY) return true;
|
|
230
|
+
let matched = 0;
|
|
231
|
+
for (let i = 0; i < data.length; i++) {
|
|
232
|
+
if (fn.v(data[i], stack)) matched++;
|
|
233
|
+
}
|
|
234
|
+
let ok = true;
|
|
235
|
+
if (matched < minC) {
|
|
236
|
+
if (errors !== NOERRORS) errors.push(err('contains', 'contains', instancePath, schemaPath + '/contains', { minContains: minC }, `must contain at least ${minC} valid item(s)`));
|
|
237
|
+
ok = false;
|
|
238
|
+
}
|
|
239
|
+
if (maxC !== undefined && matched > maxC) {
|
|
240
|
+
if (errors !== NOERRORS) errors.push(err('maxContains', 'maxContains', instancePath, schemaPath + '/maxContains', { limit: maxC }, `must NOT contain more than ${maxC} valid item(s)`));
|
|
241
|
+
ok = false;
|
|
242
|
+
}
|
|
243
|
+
return ok;
|
|
244
|
+
});
|
|
245
|
+
vsteps.push((data, stack) => {
|
|
246
|
+
if (dataBits(data) !== T_ARRAY) return true;
|
|
247
|
+
let matched = 0;
|
|
248
|
+
for (let i = 0; i < data.length; i++) {
|
|
249
|
+
if (fn.v(data[i], stack)) { matched++; if (maxC === undefined && matched >= minC) return true; }
|
|
250
|
+
}
|
|
251
|
+
return matched >= minC && (maxC === undefined || matched <= maxC);
|
|
252
|
+
});
|
|
253
|
+
}
|
|
254
|
+
|
|
255
|
+
// Objects
|
|
256
|
+
if (P.properties !== null || P.patternProperties !== null || P.additionalProperties !== undefined || P.propertyNames !== undefined) {
|
|
257
|
+
const props = P.properties;
|
|
258
|
+
const propFns = props !== null ? new Map() : null;
|
|
259
|
+
if (props !== null) {
|
|
260
|
+
for (const [key, entry] of props) propFns.set(key, { fn: compileNode(interp, entry.node, base, memo), seg: entry.seg, schemaSeg: entry.schemaSeg });
|
|
261
|
+
}
|
|
262
|
+
const patterns = P.patternProperties !== null
|
|
263
|
+
? P.patternProperties.map((e) => ({ re: e.re, src: e.src, fn: compileNode(interp, e.node, base, memo) }))
|
|
264
|
+
: null;
|
|
265
|
+
const apFn = P.additionalProperties !== undefined ? compileNode(interp, P.additionalProperties, base, memo) : null;
|
|
266
|
+
const pnFn = P.propertyNames !== undefined ? compileNode(interp, P.propertyNames, base, memo) : null;
|
|
267
|
+
steps.push((data, errors, instancePath, schemaPath, stack) => {
|
|
268
|
+
if (dataBits(data) !== T_OBJECT) return true;
|
|
269
|
+
let ok = true;
|
|
270
|
+
const keys = Object.keys(data);
|
|
271
|
+
if (pnFn !== null) {
|
|
272
|
+
for (const key of keys) {
|
|
273
|
+
if (!pnFn.c(key, errors, instancePath + '/' + escapePointer(key), schemaPath + '/propertyNames', stack)) {
|
|
274
|
+
ok = false;
|
|
275
|
+
if (errors === NOERRORS) return false;
|
|
276
|
+
}
|
|
277
|
+
}
|
|
278
|
+
}
|
|
279
|
+
for (let k = 0; k < keys.length; k++) {
|
|
280
|
+
const key = keys[k];
|
|
281
|
+
let evaluated = false;
|
|
282
|
+
const prop = propFns !== null ? propFns.get(key) : undefined;
|
|
283
|
+
if (prop !== undefined) {
|
|
284
|
+
if (!prop.fn.c(data[key], errors, instancePath + prop.seg, schemaPath + prop.schemaSeg, stack)) {
|
|
285
|
+
ok = false;
|
|
286
|
+
if (errors === NOERRORS) return false;
|
|
287
|
+
}
|
|
288
|
+
evaluated = true;
|
|
289
|
+
}
|
|
290
|
+
if (patterns !== null) {
|
|
291
|
+
for (let pi = 0; pi < patterns.length; pi++) {
|
|
292
|
+
const pp = patterns[pi];
|
|
293
|
+
if (pp.re.test(key)) {
|
|
294
|
+
if (!pp.fn.c(data[key], errors, instancePath + '/' + escapePointer(key), schemaPath + '/patternProperties/' + escapePointer(pp.src), stack)) {
|
|
295
|
+
ok = false;
|
|
296
|
+
if (errors === NOERRORS) return false;
|
|
297
|
+
}
|
|
298
|
+
evaluated = true;
|
|
299
|
+
}
|
|
300
|
+
}
|
|
301
|
+
}
|
|
302
|
+
if (!evaluated && apFn !== null) {
|
|
303
|
+
if (!apFn.c(data[key], errors, instancePath + '/' + escapePointer(key), schemaPath + '/additionalProperties', stack)) {
|
|
304
|
+
ok = false;
|
|
305
|
+
if (errors === NOERRORS) return false;
|
|
306
|
+
}
|
|
307
|
+
}
|
|
308
|
+
}
|
|
309
|
+
return ok;
|
|
310
|
+
});
|
|
311
|
+
vsteps.push((data, stack) => {
|
|
312
|
+
if (dataBits(data) !== T_OBJECT) return true;
|
|
313
|
+
const keys = Object.keys(data);
|
|
314
|
+
if (pnFn !== null) {
|
|
315
|
+
for (const key of keys) if (!pnFn.v(key, stack)) return false;
|
|
316
|
+
}
|
|
317
|
+
for (let k = 0; k < keys.length; k++) {
|
|
318
|
+
const key = keys[k];
|
|
319
|
+
let evaluated = false;
|
|
320
|
+
const prop = propFns !== null ? propFns.get(key) : undefined;
|
|
321
|
+
if (prop !== undefined) {
|
|
322
|
+
if (!prop.fn.v(data[key], stack)) return false;
|
|
323
|
+
evaluated = true;
|
|
324
|
+
}
|
|
325
|
+
if (patterns !== null) {
|
|
326
|
+
for (let pi = 0; pi < patterns.length; pi++) {
|
|
327
|
+
const pp = patterns[pi];
|
|
328
|
+
if (pp.re.test(key)) {
|
|
329
|
+
if (!pp.fn.v(data[key], stack)) return false;
|
|
330
|
+
evaluated = true;
|
|
331
|
+
}
|
|
332
|
+
}
|
|
333
|
+
}
|
|
334
|
+
if (!evaluated && apFn !== null) {
|
|
335
|
+
if (!apFn.v(data[key], stack)) return false;
|
|
336
|
+
}
|
|
337
|
+
}
|
|
338
|
+
return true;
|
|
339
|
+
});
|
|
340
|
+
}
|
|
341
|
+
if (P.dependentSchemas !== null) {
|
|
342
|
+
const entries = P.dependentSchemas.map(([key, v]) => [key, compileNode(interp, v, base, memo), escapePointer(key)]);
|
|
343
|
+
steps.push((data, errors, instancePath, schemaPath, stack) => {
|
|
344
|
+
if (dataBits(data) !== T_OBJECT) return true;
|
|
345
|
+
let ok = true;
|
|
346
|
+
for (const [key, fn, ek] of entries) {
|
|
347
|
+
if (Object.hasOwn(data, key)) {
|
|
348
|
+
if (!fn.c(data, errors, instancePath, schemaPath + '/dependentSchemas/' + ek, stack)) {
|
|
349
|
+
ok = false;
|
|
350
|
+
if (errors === NOERRORS) return false;
|
|
351
|
+
}
|
|
352
|
+
}
|
|
353
|
+
}
|
|
354
|
+
return ok;
|
|
355
|
+
});
|
|
356
|
+
vsteps.push((data, stack) => {
|
|
357
|
+
if (dataBits(data) !== T_OBJECT) return true;
|
|
358
|
+
for (const [key, fn] of entries) {
|
|
359
|
+
if (Object.hasOwn(data, key) && !fn.v(data, stack)) return false;
|
|
360
|
+
}
|
|
361
|
+
return true;
|
|
362
|
+
});
|
|
363
|
+
}
|
|
364
|
+
if (P.propertyDependencies !== null) {
|
|
365
|
+
const entries = P.propertyDependencies.map(([key, choices]) => {
|
|
366
|
+
const m = new Map();
|
|
367
|
+
for (const [value, v] of choices) m.set(value, compileNode(interp, v, base, memo));
|
|
368
|
+
return [key, m];
|
|
369
|
+
});
|
|
370
|
+
steps.push((data, errors, instancePath, schemaPath, stack) => {
|
|
371
|
+
if (dataBits(data) !== T_OBJECT) return true;
|
|
372
|
+
let ok = true;
|
|
373
|
+
for (const [key, choices] of entries) {
|
|
374
|
+
if (!Object.hasOwn(data, key)) continue;
|
|
375
|
+
const value = data[key];
|
|
376
|
+
if (typeof value !== 'string') continue;
|
|
377
|
+
const fn = choices.get(value);
|
|
378
|
+
if (fn === undefined) continue;
|
|
379
|
+
const branchPath = schemaPath + '/propertyDependencies/' + escapePointer(key) + '/' + escapePointer(value);
|
|
380
|
+
if (!fn.c(data, errors, instancePath, branchPath, stack)) {
|
|
381
|
+
ok = false;
|
|
382
|
+
if (errors === NOERRORS) return false;
|
|
383
|
+
}
|
|
384
|
+
}
|
|
385
|
+
return ok;
|
|
386
|
+
});
|
|
387
|
+
vsteps.push((data, stack) => {
|
|
388
|
+
if (dataBits(data) !== T_OBJECT) return true;
|
|
389
|
+
for (const [key, choices] of entries) {
|
|
390
|
+
if (!Object.hasOwn(data, key)) continue;
|
|
391
|
+
const value = data[key];
|
|
392
|
+
if (typeof value !== 'string') continue;
|
|
393
|
+
const fn = choices.get(value);
|
|
394
|
+
if (fn !== undefined && !fn.v(data, stack)) return false;
|
|
395
|
+
}
|
|
396
|
+
return true;
|
|
397
|
+
});
|
|
398
|
+
}
|
|
399
|
+
|
|
400
|
+
// In-place applicators
|
|
401
|
+
if (P.allOf !== null) {
|
|
402
|
+
const fns = P.allOf.map((v) => compileNode(interp, v, base, memo));
|
|
403
|
+
steps.push((data, errors, instancePath, schemaPath, stack) => {
|
|
404
|
+
let ok = true;
|
|
405
|
+
for (let i = 0; i < fns.length; i++) {
|
|
406
|
+
if (!fns[i].c(data, errors, instancePath, schemaPath + '/allOf/' + i, stack)) {
|
|
407
|
+
ok = false;
|
|
408
|
+
if (errors === NOERRORS) return false;
|
|
409
|
+
}
|
|
410
|
+
}
|
|
411
|
+
return ok;
|
|
412
|
+
});
|
|
413
|
+
vsteps.push((data, stack) => {
|
|
414
|
+
for (let i = 0; i < fns.length; i++) if (!fns[i].v(data, stack)) return false;
|
|
415
|
+
return true;
|
|
416
|
+
});
|
|
417
|
+
}
|
|
418
|
+
if (P.anyOf !== null) {
|
|
419
|
+
const fns = P.anyOf.map((v) => compileNode(interp, v, base, memo));
|
|
420
|
+
steps.push((data, errors, instancePath, schemaPath, stack) => {
|
|
421
|
+
const scratch = errors === NOERRORS ? NOERRORS : [];
|
|
422
|
+
let any = false;
|
|
423
|
+
for (let i = 0; i < fns.length; i++) {
|
|
424
|
+
if (fns[i].c(data, scratch, instancePath, schemaPath + '/anyOf/' + i, stack)) any = true;
|
|
425
|
+
}
|
|
426
|
+
if (!any) {
|
|
427
|
+
if (errors !== NOERRORS) {
|
|
428
|
+
for (const e of scratch) errors.push(e);
|
|
429
|
+
errors.push(err('anyOf', 'anyOf', instancePath, schemaPath + '/anyOf', {}, 'must match a schema in anyOf'));
|
|
430
|
+
}
|
|
431
|
+
return false;
|
|
432
|
+
}
|
|
433
|
+
return true;
|
|
434
|
+
});
|
|
435
|
+
vsteps.push((data, stack) => {
|
|
436
|
+
for (let i = 0; i < fns.length; i++) if (fns[i].v(data, stack)) return true;
|
|
437
|
+
return false;
|
|
438
|
+
});
|
|
439
|
+
}
|
|
440
|
+
if (P.oneOf !== null) {
|
|
441
|
+
const fns = P.oneOf.map((v) => compileNode(interp, v, base, memo));
|
|
442
|
+
steps.push((data, errors, instancePath, schemaPath, stack) => {
|
|
443
|
+
const scratch = errors === NOERRORS ? NOERRORS : [];
|
|
444
|
+
let count = 0;
|
|
445
|
+
for (let i = 0; i < fns.length; i++) {
|
|
446
|
+
if (fns[i].c(data, scratch, instancePath, schemaPath + '/oneOf/' + i, stack)) count++;
|
|
447
|
+
}
|
|
448
|
+
if (count !== 1) {
|
|
449
|
+
if (errors !== NOERRORS) {
|
|
450
|
+
if (count === 0) for (const e of scratch) errors.push(e);
|
|
451
|
+
errors.push(err('oneOf', 'oneOf', instancePath, schemaPath + '/oneOf', { passingSchemas: count }, 'must match exactly one schema in oneOf'));
|
|
452
|
+
}
|
|
453
|
+
return false;
|
|
454
|
+
}
|
|
455
|
+
return true;
|
|
456
|
+
});
|
|
457
|
+
vsteps.push((data, stack) => {
|
|
458
|
+
let count = 0;
|
|
459
|
+
for (let i = 0; i < fns.length; i++) {
|
|
460
|
+
if (fns[i].v(data, stack)) { count++; if (count > 1) return false; }
|
|
461
|
+
}
|
|
462
|
+
return count === 1;
|
|
463
|
+
});
|
|
464
|
+
}
|
|
465
|
+
if (P.not !== undefined) {
|
|
466
|
+
const fn = compileNode(interp, P.not, base, memo);
|
|
467
|
+
steps.push((data, errors, instancePath, schemaPath, stack) => {
|
|
468
|
+
if (fn.v(data, stack)) {
|
|
469
|
+
if (errors !== NOERRORS) errors.push(err('not', 'not', instancePath, schemaPath + '/not', {}, 'must NOT be valid'));
|
|
470
|
+
return false;
|
|
471
|
+
}
|
|
472
|
+
return true;
|
|
473
|
+
});
|
|
474
|
+
vsteps.push((data, stack) => !fn.v(data, stack));
|
|
475
|
+
}
|
|
476
|
+
if (P.if !== undefined) {
|
|
477
|
+
const ifFn = compileNode(interp, P.if, base, memo);
|
|
478
|
+
const thenFn = P.then !== undefined ? compileNode(interp, P.then, base, memo) : null;
|
|
479
|
+
const elseFn = P.else !== undefined ? compileNode(interp, P.else, base, memo) : null;
|
|
480
|
+
steps.push((data, errors, instancePath, schemaPath, stack) => {
|
|
481
|
+
if (ifFn.v(data, stack)) {
|
|
482
|
+
if (thenFn !== null) return thenFn.c(data, errors, instancePath, schemaPath + '/then', stack);
|
|
483
|
+
} else if (elseFn !== null) {
|
|
484
|
+
return elseFn.c(data, errors, instancePath, schemaPath + '/else', stack);
|
|
485
|
+
}
|
|
486
|
+
return true;
|
|
487
|
+
});
|
|
488
|
+
vsteps.push((data, stack) => {
|
|
489
|
+
if (ifFn.v(data, stack)) {
|
|
490
|
+
if (thenFn !== null) return thenFn.v(data, stack);
|
|
491
|
+
} else if (elseFn !== null) {
|
|
492
|
+
return elseFn.v(data, stack);
|
|
493
|
+
}
|
|
494
|
+
return true;
|
|
495
|
+
});
|
|
496
|
+
}
|
|
497
|
+
|
|
498
|
+
let cfn;
|
|
499
|
+
if (steps.length === 0) cfn = TRUE_FN;
|
|
500
|
+
else if (steps.length === 1) cfn = steps[0];
|
|
501
|
+
else {
|
|
502
|
+
const arr = steps;
|
|
503
|
+
cfn = (d, e, ip, sp, st) => {
|
|
504
|
+
let ok = true;
|
|
505
|
+
for (let i = 0; i < arr.length; i++) {
|
|
506
|
+
if (!arr[i](d, e, ip, sp, st)) {
|
|
507
|
+
if (e === NOERRORS) return false;
|
|
508
|
+
ok = false;
|
|
509
|
+
}
|
|
510
|
+
}
|
|
511
|
+
return ok;
|
|
512
|
+
};
|
|
513
|
+
}
|
|
514
|
+
let vfn;
|
|
515
|
+
if (vsteps.length === 0) vfn = TRUE_FN;
|
|
516
|
+
else if (vsteps.length === 1) vfn = vsteps[0];
|
|
517
|
+
else if (vsteps.length === 2) {
|
|
518
|
+
const [a, b] = vsteps;
|
|
519
|
+
vfn = (d, st) => a(d, st) && b(d, st);
|
|
520
|
+
} else {
|
|
521
|
+
const arr = vsteps;
|
|
522
|
+
vfn = (d, st) => {
|
|
523
|
+
for (let i = 0; i < arr.length; i++) if (!arr[i](d, st)) return false;
|
|
524
|
+
return true;
|
|
525
|
+
};
|
|
526
|
+
}
|
|
527
|
+
box.c = cfn;
|
|
528
|
+
box.v = vfn;
|
|
529
|
+
pair.c = cfn;
|
|
530
|
+
pair.v = vfn;
|
|
531
|
+
perBase.set(base, pair);
|
|
532
|
+
return pair;
|
|
533
|
+
}
|
|
534
|
+
|
|
535
|
+
// Entry point: returns a compiled root function or null when the schema is
|
|
536
|
+
// outside the compiler's scope.
|
|
537
|
+
function compileInterpreter(interp) {
|
|
538
|
+
if (!compilable(interp, interp.rootNode, interp.state.rootBase, new Map())) return null;
|
|
539
|
+
return compileNode(interp, interp.rootNode, interp.state.rootBase, new Map());
|
|
540
|
+
}
|
|
541
|
+
|
|
542
|
+
return { compileInterpreter };
|
|
543
|
+
}
|
|
544
|
+
|
|
545
|
+
module.exports = { install };
|
package/lib/safe-regex-source.js
CHANGED
|
@@ -5,4 +5,4 @@
|
|
|
5
5
|
// into standalone output without a runtime `fs` read. Kept in sync by
|
|
6
6
|
// `tests/test_safe_regex_source_sync.js`.
|
|
7
7
|
|
|
8
|
-
module.exports = "'use strict'\n\n// Linear-time regex engine for JSON Schema `pattern`, used in place of JS RegExp\n// so an adversarial input cannot trigger catastrophic backtracking (ReDoS).\n//\n// It is a Pike VM: the pattern compiles to a small instruction program, and the\n// VM simulates all NFA threads in lockstep over the input, deduping by program\n// counter. Runtime is O(input * program), with no backtracking.\n//\n// Supported (the RE2 subset, which is what ata's native path also accepts):\n// literals, ., character classes, \\d \\w \\s \\D \\W \\S, anchors ^ $, quantifiers\n// * + ? {n} {n,} {n,m} (greedy or lazy, same language for a boolean test),\n// groups ( ) (?: ), alternation |. Backreferences and lookaround are not\n// supported by linear engines; compileSafe throws on them so the caller can\n// decide (ata's codegen rejects such schemas rather than risk a hang).\n\nconst WS = [[9, 13], [32, 32], [160, 160]]\nconst DIGIT = [[48, 57]]\nconst WORD = [[48, 57], [65, 90], [97, 122], [95, 95]]\n\nfunction parse (src) {\n let i = 0\n const len = src.length\n const peek = () => src[i]\n const eof = () => i >= len\n\n function parseAlt () {\n const opts = [parseConcat()]\n while (!eof() && peek() === '|') { i++; opts.push(parseConcat()) }\n return opts.length === 1 ? opts[0] : { t: 'alt', opts }\n }\n\n function parseConcat () {\n const parts = []\n while (!eof() && peek() !== '|' && peek() !== ')') parts.push(parseRepeat())\n if (parts.length === 0) return { t: 'empty' }\n return parts.length === 1 ? parts[0] : { t: 'concat', parts }\n }\n\n function parseRepeat () {\n let node = parseAtom()\n while (!eof()) {\n const ch = peek()\n if (ch === '*') { i++; node = { t: 'star', child: node } }\n else if (ch === '+') { i++; node = { t: 'plus', child: node } }\n else if (ch === '?') { i++; node = { t: 'quest', child: node } }\n else if (ch === '{') {\n const saved = i\n const q = tryQuantifier()\n if (!q) { i = saved; break }\n node = { t: 'repeat', child: node, min: q.min, max: q.max }\n } else break\n // a trailing ? makes the quantifier lazy; same language for a boolean test\n if (!eof() && peek() === '?') i++\n }\n return node\n }\n\n function tryQuantifier () {\n // assumes current char is '{'\n i++\n let min = ''\n while (!eof() && /[0-9]/.test(peek())) { min += peek(); i++ }\n if (min === '') return null\n let max\n if (peek() === '}') { i++; return { min: +min, max: +min } }\n if (peek() === ',') {\n i++\n let m = ''\n while (!eof() && /[0-9]/.test(peek())) { m += peek(); i++ }\n if (peek() !== '}') return null\n i++\n max = m === '' ? Infinity : +m\n return { min: +min, max }\n }\n return null\n }\n\n function parseAtom () {\n const ch = peek()\n if (ch === '(') {\n i++\n if (src[i] === '?') {\n if (src[i + 1] === ':') { i += 2 }\n else throw new Error('unsupported group (lookaround/named) in pattern')\n }\n const child = parseAlt()\n if (peek() !== ')') throw new Error('unbalanced ( in pattern')\n i++\n return { t: 'group', child }\n }\n if (ch === '[') return parseClass()\n if (ch === '.') { i++; return { t: 'any' } }\n if (ch === '^') { i++; return { t: 'bol' } }\n if (ch === '$') { i++; return { t: 'eol' } }\n if (ch === '\\\\') return parseEscape(false)\n if (ch === ')' || ch === '|') return { t: 'empty' }\n i++\n return { t: 'char', c: ch.charCodeAt(0) }\n }\n\n function parseClass () {\n i++ // [\n let neg = false\n if (peek() === '^') { neg = true; i++ }\n const ranges = []\n while (!eof() && peek() !== ']') {\n let lo\n if (peek() === '\\\\') {\n const esc = parseEscape(true)\n if (esc.t === 'classpart') { for (const r of esc.ranges) ranges.push(r); continue }\n lo = esc.c\n } else { lo = peek().charCodeAt(0); i++ }\n if (peek() === '-' && src[i + 1] !== ']' && i + 1 < len) {\n i++ // -\n let hi\n if (peek() === '\\\\') { const e = parseEscape(true); hi = e.c } else { hi = peek().charCodeAt(0); i++ }\n ranges.push([lo, hi])\n } else {\n ranges.push([lo, lo])\n }\n }\n if (peek() !== ']') throw new Error('unbalanced [ in pattern')\n i++\n return { t: 'class', neg, ranges }\n }\n\n function parseEscape (inClass) {\n i++ // backslash\n if (eof()) throw new Error('trailing backslash in pattern')\n const ch = peek(); i++\n switch (ch) {\n case 'd': return inClass ? { t: 'classpart', ranges: DIGIT } : { t: 'class', neg: false, ranges: DIGIT }\n case 'w': return inClass ? { t: 'classpart', ranges: WORD } : { t: 'class', neg: false, ranges: WORD }\n case 's': return inClass ? { t: 'classpart', ranges: WS } : { t: 'class', neg: false, ranges: WS }\n case 'D': if (inClass) throw new Error('\\\\D inside a class is not supported'); return { t: 'class', neg: true, ranges: DIGIT }\n case 'W': if (inClass) throw new Error('\\\\W inside a class is not supported'); return { t: 'class', neg: true, ranges: WORD }\n case 'S': if (inClass) throw new Error('\\\\S inside a class is not supported'); return { t: 'class', neg: true, ranges: WS }\n case 'n': return { t: 'char', c: 10 }\n case 'r': return { t: 'char', c: 13 }\n case 't': return { t: 'char', c: 9 }\n case 'f': return { t: 'char', c: 12 }\n case 'v': return { t: 'char', c: 11 }\n case '0': return { t: 'char', c: 0 }\n case 'x': { const h = src.slice(i, i + 2); i += 2; return { t: 'char', c: parseInt(h, 16) } }\n case 'u': { const h = src.slice(i, i + 4); i += 4; return { t: 'char', c: parseInt(h, 16) } }\n case 'b': if (inClass) return { t: 'char', c: 8 }; throw new Error('\\\\b word boundary is not supported')\n default:\n if (/[1-9]/.test(ch)) throw new Error('backreferences are not supported in pattern')\n return { t: 'char', c: ch.charCodeAt(0) }\n }\n }\n\n const ast = parseAlt()\n if (!eof()) throw new Error('unexpected \"' + peek() + '\" in pattern')\n return ast\n}\n\nfunction compileProg (ast) {\n const prog = []\n const emit = (op, extra) => { const idx = prog.length; prog.push(Object.assign({ op }, extra)); return idx }\n\n function rec (n) {\n switch (n.t) {\n case 'empty': break\n case 'char': emit('char', { c: n.c }); break\n case 'any': emit('any'); break\n case 'class': emit('class', { neg: n.neg, ranges: n.ranges }); break\n case 'bol': emit('bol'); break\n case 'eol': emit('eol'); break\n case 'group': rec(n.child); break\n case 'concat': for (const p of n.parts) rec(p); break\n case 'alt': {\n const jmps = []\n for (let k = 0; k < n.opts.length; k++) {\n if (k < n.opts.length - 1) {\n const sp = emit('split', { x: 0, y: 0 })\n prog[sp].x = prog.length\n rec(n.opts[k])\n jmps.push(emit('jmp', { x: 0 }))\n prog[sp].y = prog.length\n } else {\n rec(n.opts[k])\n }\n }\n for (const j of jmps) prog[j].x = prog.length\n break\n }\n case 'star': {\n const sp = emit('split', { x: 0, y: 0 })\n prog[sp].x = prog.length\n rec(n.child)\n emit('jmp', { x: sp })\n prog[sp].y = prog.length\n break\n }\n case 'plus': {\n const start = prog.length\n rec(n.child)\n const sp = emit('split', { x: start, y: 0 })\n prog[sp].y = prog.length\n break\n }\n case 'quest': {\n const sp = emit('split', { x: 0, y: 0 })\n prog[sp].x = prog.length\n rec(n.child)\n prog[sp].y = prog.length\n break\n }\n case 'repeat': {\n for (let k = 0; k < n.min; k++) rec(n.child)\n if (n.max === Infinity) {\n if (n.min === 0) rec({ t: 'star', child: n.child })\n else rec({ t: 'star', child: n.child })\n } else {\n for (let k = 0; k < n.max - n.min; k++) rec({ t: 'quest', child: n.child })\n }\n break\n }\n }\n }\n\n rec(ast)\n emit('match')\n return prog\n}\n\nfunction matchClass (instr, c) {\n let inside = false\n const r = instr.ranges\n for (let k = 0; k < r.length; k++) { if (c >= r[k][0] && c <= r[k][1]) { inside = true; break } }\n return instr.neg ? !inside : inside\n}\n\nfunction makeRunner (prog) {\n const n = prog.length\n const lastGen = new Int32Array(n).fill(-1)\n let gen = 0\n const stack = []\n\n function addThread (list, pc, pos, len) {\n stack.length = 0\n stack.push(pc)\n while (stack.length) {\n const p = stack.pop()\n if (lastGen[p] === gen) continue\n lastGen[p] = gen\n const I = prog[p]\n switch (I.op) {\n case 'jmp': stack.push(I.x); break\n case 'split': stack.push(I.y); stack.push(I.x); break\n case 'bol': if (pos === 0) stack.push(p + 1); break\n case 'eol': if (pos === len) stack.push(p + 1); break\n default: list.push(p)\n }\n }\n }\n\n return function test (s) {\n const len = s.length\n let clist = []\n let nlist = []\n gen++\n addThread(clist, 0, 0, len)\n for (let pos = 0; pos <= len; pos++) {\n const c = pos < len ? s.charCodeAt(pos) : -1\n gen++\n nlist.length = 0\n for (let k = 0; k < clist.length; k++) {\n const pc = clist[k]\n const I = prog[pc]\n if (I.op === 'match') return true\n else if (I.op === 'char') { if (c === I.c) addThread(nlist, pc + 1, pos + 1, len) }\n else if (I.op === 'any') { if (c !== -1 && c !== 10) addThread(nlist, pc + 1, pos + 1, len) }\n else if (I.op === 'class') { if (c !== -1 && matchClass(I, c)) addThread(nlist, pc + 1, pos + 1, len) }\n }\n if (pos < len) addThread(nlist, 0, pos + 1, len)\n const tmp = clist; clist = nlist; nlist = tmp\n }\n return false\n }\n}\n\nfunction compileSafe (pattern) {\n const prog = compileProg(parse(pattern))\n const runner = makeRunner(prog)\n // `__ataSafe` brands the result so the standalone serializer can tell a safe\n // matcher apart from a RegExp and emit `__ataSafeRe(source)` instead.\n return { test: runner, source: pattern, __ataSafe: true }\n}\n\n// True when the linear engine can represent `src`. Used by the codegen to decide\n// between the safe matcher and a JS RegExp fallback for patterns outside the\n// supported (RE2) subset (backreferences, lookaround, etc.).\nfunction patternIsSafe (src) {\n try { compileSafe(src); return true } catch { return false }\n}\n\nmodule.exports = { compileSafe, patternIsSafe }\n";
|
|
8
|
+
module.exports = "'use strict'\n\n// Linear-time regex engine for JSON Schema `pattern`, used in place of JS RegExp\n// so an adversarial input cannot trigger catastrophic backtracking (ReDoS).\n//\n// It is a Pike VM: the pattern compiles to a small instruction program, and the\n// VM simulates all NFA threads in lockstep over the input, deduping by program\n// counter. Runtime is O(input * program), with no backtracking.\n//\n// Supported (the RE2 subset, which is what ata's native path also accepts):\n// literals, ., character classes, \\d \\w \\s \\D \\W \\S, anchors ^ $, quantifiers\n// * + ? {n} {n,} {n,m} (greedy or lazy, same language for a boolean test),\n// groups ( ) (?: ), alternation |. Backreferences and lookaround are not\n// supported by linear engines; compileSafe throws on them so the caller can\n// decide (ata's codegen rejects such schemas rather than risk a hang).\n\nconst WS = [[9, 13], [32, 32], [160, 160]]\nconst DIGIT = [[48, 57]]\nconst WORD = [[48, 57], [65, 90], [97, 122], [95, 95]]\n\nfunction parse (src) {\n let i = 0\n const len = src.length\n const peek = () => src[i]\n const eof = () => i >= len\n\n function parseAlt () {\n const opts = [parseConcat()]\n while (!eof() && peek() === '|') { i++; opts.push(parseConcat()) }\n return opts.length === 1 ? opts[0] : { t: 'alt', opts }\n }\n\n function parseConcat () {\n const parts = []\n while (!eof() && peek() !== '|' && peek() !== ')') parts.push(parseRepeat())\n if (parts.length === 0) return { t: 'empty' }\n return parts.length === 1 ? parts[0] : { t: 'concat', parts }\n }\n\n function parseRepeat () {\n let node = parseAtom()\n while (!eof()) {\n const ch = peek()\n if (ch === '*') { i++; node = { t: 'star', child: node } }\n else if (ch === '+') { i++; node = { t: 'plus', child: node } }\n else if (ch === '?') { i++; node = { t: 'quest', child: node } }\n else if (ch === '{') {\n const saved = i\n const q = tryQuantifier()\n if (!q) { i = saved; break }\n node = { t: 'repeat', child: node, min: q.min, max: q.max }\n } else break\n // a trailing ? makes the quantifier lazy; same language for a boolean test\n if (!eof() && peek() === '?') i++\n }\n return node\n }\n\n function tryQuantifier () {\n // assumes current char is '{'\n i++\n let min = ''\n while (!eof() && /[0-9]/.test(peek())) { min += peek(); i++ }\n if (min === '') return null\n let max\n if (peek() === '}') { i++; return { min: +min, max: +min } }\n if (peek() === ',') {\n i++\n let m = ''\n while (!eof() && /[0-9]/.test(peek())) { m += peek(); i++ }\n if (peek() !== '}') return null\n i++\n max = m === '' ? Infinity : +m\n return { min: +min, max }\n }\n return null\n }\n\n function parseAtom () {\n const ch = peek()\n if (ch === '(') {\n i++\n if (src[i] === '?') {\n if (src[i + 1] === ':') { i += 2 }\n else throw new Error('unsupported group (lookaround/named) in pattern')\n }\n const child = parseAlt()\n if (peek() !== ')') throw new Error('unbalanced ( in pattern')\n i++\n return { t: 'group', child }\n }\n if (ch === '[') return parseClass()\n if (ch === '.') { i++; return { t: 'any' } }\n if (ch === '^') { i++; return { t: 'bol' } }\n if (ch === '$') { i++; return { t: 'eol' } }\n if (ch === '\\\\') return parseEscape(false)\n if (ch === ')' || ch === '|') return { t: 'empty' }\n i++\n return { t: 'char', c: ch.charCodeAt(0) }\n }\n\n function parseClass () {\n i++ // [\n let neg = false\n if (peek() === '^') { neg = true; i++ }\n const ranges = []\n while (!eof() && peek() !== ']') {\n let lo\n if (peek() === '\\\\') {\n const esc = parseEscape(true)\n if (esc.t === 'classpart') { for (const r of esc.ranges) ranges.push(r); continue }\n lo = esc.c\n } else { lo = peek().charCodeAt(0); i++ }\n if (peek() === '-' && src[i + 1] !== ']' && i + 1 < len) {\n i++ // -\n let hi\n if (peek() === '\\\\') { const e = parseEscape(true); hi = e.c } else { hi = peek().charCodeAt(0); i++ }\n ranges.push([lo, hi])\n } else {\n ranges.push([lo, lo])\n }\n }\n if (peek() !== ']') throw new Error('unbalanced [ in pattern')\n i++\n return { t: 'class', neg, ranges }\n }\n\n function parseEscape (inClass) {\n i++ // backslash\n if (eof()) throw new Error('trailing backslash in pattern')\n const ch = peek(); i++\n switch (ch) {\n case 'd': return inClass ? { t: 'classpart', ranges: DIGIT } : { t: 'class', neg: false, ranges: DIGIT }\n case 'w': return inClass ? { t: 'classpart', ranges: WORD } : { t: 'class', neg: false, ranges: WORD }\n case 's': return inClass ? { t: 'classpart', ranges: WS } : { t: 'class', neg: false, ranges: WS }\n case 'D': if (inClass) throw new Error('\\\\D inside a class is not supported'); return { t: 'class', neg: true, ranges: DIGIT }\n case 'W': if (inClass) throw new Error('\\\\W inside a class is not supported'); return { t: 'class', neg: true, ranges: WORD }\n case 'S': if (inClass) throw new Error('\\\\S inside a class is not supported'); return { t: 'class', neg: true, ranges: WS }\n case 'n': return { t: 'char', c: 10 }\n case 'r': return { t: 'char', c: 13 }\n case 't': return { t: 'char', c: 9 }\n case 'f': return { t: 'char', c: 12 }\n case 'v': return { t: 'char', c: 11 }\n case '0': return { t: 'char', c: 0 }\n case 'x': { const h = src.slice(i, i + 2); i += 2; return { t: 'char', c: parseInt(h, 16) } }\n case 'u': { const h = src.slice(i, i + 4); i += 4; return { t: 'char', c: parseInt(h, 16) } }\n case 'b': if (inClass) return { t: 'char', c: 8 }; throw new Error('\\\\b word boundary is not supported')\n default:\n if (/[1-9]/.test(ch)) throw new Error('backreferences are not supported in pattern')\n return { t: 'char', c: ch.charCodeAt(0) }\n }\n }\n\n const ast = parseAlt()\n if (!eof()) throw new Error('unexpected \"' + peek() + '\" in pattern')\n return ast\n}\n\nfunction compileProg (ast) {\n const prog = []\n const emit = (op, extra) => { const idx = prog.length; prog.push(Object.assign({ op }, extra)); return idx }\n\n function rec (n) {\n switch (n.t) {\n case 'empty': break\n case 'char': emit('char', { c: n.c }); break\n case 'any': emit('any'); break\n case 'class': emit('class', { neg: n.neg, ranges: n.ranges }); break\n case 'bol': emit('bol'); break\n case 'eol': emit('eol'); break\n case 'group': rec(n.child); break\n case 'concat': for (const p of n.parts) rec(p); break\n case 'alt': {\n const jmps = []\n for (let k = 0; k < n.opts.length; k++) {\n if (k < n.opts.length - 1) {\n const sp = emit('split', { x: 0, y: 0 })\n prog[sp].x = prog.length\n rec(n.opts[k])\n jmps.push(emit('jmp', { x: 0 }))\n prog[sp].y = prog.length\n } else {\n rec(n.opts[k])\n }\n }\n for (const j of jmps) prog[j].x = prog.length\n break\n }\n case 'star': {\n const sp = emit('split', { x: 0, y: 0 })\n prog[sp].x = prog.length\n rec(n.child)\n emit('jmp', { x: sp })\n prog[sp].y = prog.length\n break\n }\n case 'plus': {\n const start = prog.length\n rec(n.child)\n const sp = emit('split', { x: start, y: 0 })\n prog[sp].y = prog.length\n break\n }\n case 'quest': {\n const sp = emit('split', { x: 0, y: 0 })\n prog[sp].x = prog.length\n rec(n.child)\n prog[sp].y = prog.length\n break\n }\n case 'repeat': {\n for (let k = 0; k < n.min; k++) rec(n.child)\n if (n.max === Infinity) {\n if (n.min === 0) rec({ t: 'star', child: n.child })\n else rec({ t: 'star', child: n.child })\n } else {\n for (let k = 0; k < n.max - n.min; k++) rec({ t: 'quest', child: n.child })\n }\n break\n }\n }\n }\n\n rec(ast)\n emit('match')\n return prog\n}\n\n// Numeric opcodes for the runner. The program is compiled once into flat\n// typed arrays so the inner loop does no property lookups or string compares.\nconst OP_CHAR = 0\nconst OP_ANY = 1\nconst OP_CLASS = 2\nconst OP_SPLIT = 3\nconst OP_JMP = 4\nconst OP_BOL = 5\nconst OP_EOL = 6\nconst OP_MATCH = 7\n\nfunction classMatcher (instr) {\n // ASCII is answered from a bitmap; anything above 0x7f walks the ranges.\n const bits = new Uint8Array(128)\n const r = instr.ranges\n for (let k = 0; k < r.length; k++) {\n const hi = Math.min(r[k][1], 127)\n for (let c = r[k][0]; c <= hi; c++) bits[c] = 1\n }\n return { bits, ranges: r, neg: instr.neg }\n}\n\nfunction matchClass (cls, c) {\n let inside\n if (c < 128) {\n inside = cls.bits[c] === 1\n } else {\n inside = false\n const r = cls.ranges\n for (let k = 0; k < r.length; k++) { if (c >= r[k][0] && c <= r[k][1]) { inside = true; break } }\n }\n return cls.neg ? !inside : inside\n}\n\nfunction makeRunner (prog) {\n const n = prog.length\n const ops = new Uint8Array(n)\n const xs = new Int32Array(n)\n const ys = new Int32Array(n)\n const cs = new Int32Array(n)\n const classes = new Array(n)\n for (let i = 0; i < n; i++) {\n const I = prog[i]\n switch (I.op) {\n case 'char': ops[i] = OP_CHAR; cs[i] = I.c; break\n case 'any': ops[i] = OP_ANY; break\n case 'class': ops[i] = OP_CLASS; classes[i] = classMatcher(I); break\n case 'split': ops[i] = OP_SPLIT; xs[i] = I.x; ys[i] = I.y; break\n case 'jmp': ops[i] = OP_JMP; xs[i] = I.x; break\n case 'bol': ops[i] = OP_BOL; break\n case 'eol': ops[i] = OP_EOL; break\n case 'match': ops[i] = OP_MATCH; break\n }\n }\n\n const lastGen = new Int32Array(n).fill(-1)\n let gen = 0\n // Each unvisited instruction is popped once and pushes at most two, so the\n // stack never holds more than 2n + 1 entries.\n const stack = new Int32Array(2 * n + 2)\n // Thread lists hold at most one entry per instruction per step.\n let clist = new Int32Array(n)\n let nlist = new Int32Array(n)\n let clen = 0\n let nlen = 0\n\n // Follows epsilon edges from `pc` and records every consuming instruction\n // (or match) reached in `list`. `lastGen` dedupes per step.\n function addThread (list, len0, pc, pos, len) {\n // Most transitions land directly on a consuming instruction; skip the\n // stack walk for those.\n if (ops[pc] <= OP_CLASS || ops[pc] === OP_MATCH) {\n if (lastGen[pc] === gen) return len0\n lastGen[pc] = gen\n list[len0] = pc\n return len0 + 1\n }\n let sp = 0\n stack[sp++] = pc\n let count = len0\n while (sp > 0) {\n const p = stack[--sp]\n if (lastGen[p] === gen) continue\n lastGen[p] = gen\n switch (ops[p]) {\n case OP_JMP: stack[sp++] = xs[p]; break\n case OP_SPLIT: stack[sp++] = ys[p]; stack[sp++] = xs[p]; break\n case OP_BOL: if (pos === 0) stack[sp++] = p + 1; break\n case OP_EOL: if (pos === len) stack[sp++] = p + 1; break\n default: list[count++] = p\n }\n }\n return count\n }\n\n // A pattern is anchored when starting it anywhere but position 0 yields no\n // thread, which is the case for `^...` and its alternations. The probe sits\n // at position 1 of a length-1 string so that only `^` can fail. For anchored\n // patterns the per-position restart below is skipped, and an empty thread\n // list means the match has already failed.\n gen++\n const anchored = addThread(nlist, 0, 0, 1, 1) === 0\n\n function testNFA (s) {\n const len = s.length\n gen++\n clen = addThread(clist, 0, 0, 0, len)\n for (let pos = 0; pos <= len; pos++) {\n const c = pos < len ? s.charCodeAt(pos) : -1\n gen++\n nlen = 0\n for (let k = 0; k < clen; k++) {\n const pc = clist[k]\n switch (ops[pc]) {\n case OP_MATCH: return true\n case OP_CHAR: if (c === cs[pc]) nlen = addThread(nlist, nlen, pc + 1, pos + 1, len); break\n case OP_ANY: if (c !== -1 && c !== 10) nlen = addThread(nlist, nlen, pc + 1, pos + 1, len); break\n case OP_CLASS: if (c !== -1 && matchClass(classes[pc], c)) nlen = addThread(nlist, nlen, pc + 1, pos + 1, len); break\n }\n }\n if (pos < len) {\n if (!anchored) nlen = addThread(nlist, nlen, 0, pos + 1, len)\n else if (nlen === 0) return false\n }\n const tmp = clist; clist = nlist; nlist = tmp\n clen = nlen\n }\n return false\n }\n\n // Lazy DFA on top of the NFA. A DFA state is the set of consuming\n // instructions live at a position; transitions are computed on first use\n // and cached per ASCII character. `^` and `$` depend on position, so the\n // closure is taken with flags for \"at start\" and \"at end\", which gives two\n // start states and two transition tables per state. The state count is\n // capped; past the cap the matcher falls back to the NFA walk above, so the\n // time bound stays linear either way.\n const MAX_STATES = 256\n const states = []\n const stateIds = new Map()\n let overflow = false\n\n function closure (list, count, pc, atStart, atEnd) {\n // Same walk as addThread, with the position replaced by the two flags.\n let sp = 0\n stack[sp++] = pc\n while (sp > 0) {\n const p = stack[--sp]\n if (lastGen[p] === gen) continue\n lastGen[p] = gen\n switch (ops[p]) {\n case OP_JMP: stack[sp++] = xs[p]; break\n case OP_SPLIT: stack[sp++] = ys[p]; stack[sp++] = xs[p]; break\n case OP_BOL: if (atStart) stack[sp++] = p + 1; break\n case OP_EOL: if (atEnd) stack[sp++] = p + 1; break\n default: list[count++] = p\n }\n }\n return count\n }\n\n function internState (list, count) {\n const pcs = Array.from(list.subarray(0, count)).sort((a, b) => a - b)\n const key = pcs.join(',')\n let id = stateIds.get(key)\n if (id !== undefined) return id\n if (states.length >= MAX_STATES) { overflow = true; return -1 }\n id = states.length\n let isMatch = false\n for (let k = 0; k < pcs.length; k++) if (ops[pcs[k]] === OP_MATCH) { isMatch = true; break }\n states.push({ pcs: Int32Array.from(pcs), isMatch, next: new Int32Array(128).fill(-2), nextEnd: new Int32Array(128).fill(-2) })\n stateIds.set(key, id)\n return id\n }\n\n function step (state, c, atEnd) {\n gen++\n let count = 0\n const pcs = state.pcs\n for (let k = 0; k < pcs.length; k++) {\n const pc = pcs[k]\n switch (ops[pc]) {\n case OP_CHAR: if (c === cs[pc]) count = closure(nlist, count, pc + 1, false, atEnd); break\n case OP_ANY: if (c !== 10) count = closure(nlist, count, pc + 1, false, atEnd); break\n case OP_CLASS: if (matchClass(classes[pc], c)) count = closure(nlist, count, pc + 1, false, atEnd); break\n }\n }\n if (!anchored) count = closure(nlist, count, 0, false, atEnd)\n return internState(nlist, count)\n }\n\n let startEmpty = -2\n let startNonEmpty = -2\n\n function startState (atEnd) {\n gen++\n const count = closure(nlist, 0, 0, true, atEnd)\n return internState(nlist, count)\n }\n\n function testDFA (s) {\n const len = s.length\n let id\n if (len === 0) {\n if (startEmpty === -2) startEmpty = startState(true)\n id = startEmpty\n } else {\n if (startNonEmpty === -2) startNonEmpty = startState(false)\n id = startNonEmpty\n }\n if (id < 0) return testNFA(s)\n let state = states[id]\n for (let pos = 0; pos < len; pos++) {\n if (state.isMatch) return true\n const c = s.charCodeAt(pos)\n const atEnd = pos + 1 === len\n let nid\n if (c < 128) {\n const table = atEnd ? state.nextEnd : state.next\n nid = table[c]\n if (nid === -2) { nid = step(state, c, atEnd); table[c] = nid }\n } else {\n nid = step(state, c, atEnd)\n }\n if (nid < 0) return testNFA(s)\n state = states[nid]\n if (anchored && state.pcs.length === 0) return false\n }\n return state.isMatch\n }\n\n return function test (s) {\n return overflow ? testNFA(s) : testDFA(s)\n }\n}\n\nfunction compileSafe (pattern) {\n const prog = compileProg(parse(pattern))\n const runner = makeRunner(prog)\n // `__ataSafe` brands the result so the standalone serializer can tell a safe\n // matcher apart from a RegExp and emit `__ataSafeRe(source)` instead.\n return { test: runner, source: pattern, __ataSafe: true }\n}\n\n// True when the linear engine can represent `src`. Used by the codegen to decide\n// between the safe matcher and a JS RegExp fallback for patterns outside the\n// supported (RE2) subset (backreferences, lookaround, etc.).\nfunction patternIsSafe (src) {\n try { compileSafe(src); return true } catch { return false }\n}\n\nmodule.exports = { compileSafe, patternIsSafe }\n";
|