ata-validator 0.17.2 → 0.17.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +6 -0
- package/index.js +36 -6
- package/lib/js-compiler.js +63 -18
- package/lib/safe-regex.js +300 -0
- package/package.json +2 -2
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,12 @@
|
|
|
2
2
|
|
|
3
3
|
All notable changes to ata-validator are documented here. The format follows [Keep a Changelog](https://keepachangelog.com/), and this project adheres to semantic versioning.
|
|
4
4
|
|
|
5
|
+
## 0.17.3 - 2026-05-25
|
|
6
|
+
|
|
7
|
+
### Security
|
|
8
|
+
|
|
9
|
+
- User-supplied `pattern`, `patternProperties`, and `propertyNames` regexes now run through a linear-time matching engine, so a crafted schema or input can no longer trigger catastrophic backtracking (ReDoS). Patterns the engine cannot represent, such as those using backreferences, fall back to the native `RegExp`. The built-in `format` checks (`email`, `uri`, `uri-reference`, `hostname`, `ipv4`, `ipv6`, `date`, `date-time`, `time`, `duration`, `uuid`) were routed through the same engine and stay linear on adversarial input.
|
|
10
|
+
|
|
5
11
|
## 0.17.2 - 2026-05-24
|
|
6
12
|
|
|
7
13
|
### Fixed
|
package/index.js
CHANGED
|
@@ -299,6 +299,27 @@ const _CP_LEN_SOURCE = `function _cpLen(s) {
|
|
|
299
299
|
return len;
|
|
300
300
|
}`;
|
|
301
301
|
|
|
302
|
+
// The linear-time regex engine, inlined verbatim into standalone output so a
|
|
303
|
+
// compiled module that uses safe `pattern` matchers has no runtime dependency on
|
|
304
|
+
// ata-validator. Read from the source file (one copy, no drift) with the strict
|
|
305
|
+
// directive and CommonJS exports stripped, plus the `__ataSafeRe` alias the
|
|
306
|
+
// emitted code calls. The engine has no eval/new Function, so the embed is
|
|
307
|
+
// CSP-safe.
|
|
308
|
+
const SAFE_REGEX_EMBED = (() => {
|
|
309
|
+
const raw = require("fs").readFileSync(require("path").join(__dirname, "lib", "safe-regex.js"), "utf8");
|
|
310
|
+
const body = raw
|
|
311
|
+
.replace(/^'use strict'\s*\n/, "")
|
|
312
|
+
.replace(/\nmodule\.exports[^\n]*\n?/, "\n");
|
|
313
|
+
return body.trimEnd() + "\nconst __ataSafeRe = compileSafe;";
|
|
314
|
+
})();
|
|
315
|
+
|
|
316
|
+
// Returns the engine embed when any supplied compiled function references the
|
|
317
|
+
// safe matcher (jsFn._usesSafeRe), else an empty string so non-pattern modules
|
|
318
|
+
// pay zero bytes.
|
|
319
|
+
function safeRePrelude(...fns) {
|
|
320
|
+
return fns.some((f) => f && f._usesSafeRe) ? SAFE_REGEX_EMBED + "\n" : "";
|
|
321
|
+
}
|
|
322
|
+
|
|
302
323
|
// Above this size, simdjson On Demand (selective field access) beats JSON.parse
|
|
303
324
|
// (which must materialize the full JS object tree). Buffer.from + NAPI ~2x faster.
|
|
304
325
|
const SIMDJSON_THRESHOLD = 8192;
|
|
@@ -1152,7 +1173,7 @@ class Validator {
|
|
|
1152
1173
|
return `// Auto-generated by ata-validator — do not edit
|
|
1153
1174
|
'use strict';
|
|
1154
1175
|
${_CP_LEN_SOURCE}
|
|
1155
|
-
${preambleSrc}
|
|
1176
|
+
${safeRePrelude(jsFn, jsErrFn)}${preambleSrc}
|
|
1156
1177
|
const boolFn = function(d) {
|
|
1157
1178
|
${src}
|
|
1158
1179
|
};
|
|
@@ -1188,8 +1209,9 @@ module.exports = { boolFn, hybridFactory, errFn };
|
|
|
1188
1209
|
const src = jsFn._source;
|
|
1189
1210
|
|
|
1190
1211
|
let errCore = '';
|
|
1212
|
+
let jsErrFn = null;
|
|
1191
1213
|
if (!abortEarly) {
|
|
1192
|
-
|
|
1214
|
+
jsErrFn = compileToJSCodegenWithErrors(
|
|
1193
1215
|
typeof this._schemaObj === 'object' ? this._schemaObj : {},
|
|
1194
1216
|
null,
|
|
1195
1217
|
undefined,
|
|
@@ -1220,7 +1242,9 @@ module.exports = { boolFn, hybridFactory, errFn };
|
|
|
1220
1242
|
lines.push(`const ${name} = ${JSON.stringify(val)};`);
|
|
1221
1243
|
continue;
|
|
1222
1244
|
}
|
|
1223
|
-
if (val
|
|
1245
|
+
if (val && val.__ataSafe) {
|
|
1246
|
+
lines.push(`const ${name} = __ataSafeRe(${JSON.stringify(val.source)});`);
|
|
1247
|
+
} else if (val instanceof RegExp) {
|
|
1224
1248
|
const flags = val.flags;
|
|
1225
1249
|
lines.push(`const ${name} = new RegExp(${JSON.stringify(val.source)}${flags ? ', ' + JSON.stringify(flags) : ''});`);
|
|
1226
1250
|
} else if (val instanceof Set) {
|
|
@@ -1254,7 +1278,7 @@ module.exports = { boolFn, hybridFactory, errFn };
|
|
|
1254
1278
|
// Schema is embedded; runtime has zero dependency on ata-validator.
|
|
1255
1279
|
'use strict';
|
|
1256
1280
|
${_CP_LEN_SOURCE}
|
|
1257
|
-
${schemaSourceConst}const VALID = Object.freeze({ valid: true, errors: Object.freeze([]) });
|
|
1281
|
+
${safeRePrelude(jsFn, jsErrFn)}${schemaSourceConst}const VALID = Object.freeze({ valid: true, errors: Object.freeze([]) });
|
|
1258
1282
|
const ABORT = Object.freeze({
|
|
1259
1283
|
valid: false,
|
|
1260
1284
|
errors: Object.freeze([Object.freeze({
|
|
@@ -1490,6 +1514,7 @@ Validator.bundleStandalone = function (schemas, opts) {
|
|
|
1490
1514
|
const bundleOpts = haveIds ? { ...(opts || {}), schemas } : (opts || {});
|
|
1491
1515
|
const format = (opts && opts.format) || 'cjs';
|
|
1492
1516
|
const R = "Object.freeze({valid:true,errors:Object.freeze([])})";
|
|
1517
|
+
let bundleUsesSafeRe = false;
|
|
1493
1518
|
const fns = schemas.map((schema) => {
|
|
1494
1519
|
const v = new Validator(schema, bundleOpts);
|
|
1495
1520
|
v._ensureCompiled();
|
|
@@ -1500,6 +1525,7 @@ Validator.bundleStandalone = function (schemas, opts) {
|
|
|
1500
1525
|
v._schemaMap,
|
|
1501
1526
|
v._userFormats,
|
|
1502
1527
|
);
|
|
1528
|
+
if (jsFn._usesSafeRe || (jsErrFn && jsErrFn._usesSafeRe)) bundleUsesSafeRe = true;
|
|
1503
1529
|
const errBody =
|
|
1504
1530
|
jsErrFn && jsErrFn._errSource
|
|
1505
1531
|
? jsErrFn._errSource
|
|
@@ -1525,10 +1551,11 @@ Validator.bundleStandalone = function (schemas, opts) {
|
|
|
1525
1551
|
return `(function(R){${preamble}var E=function(d){var _all=true;${errBody}};return function(d){${jsFn._hybridSource}}})(R)`;
|
|
1526
1552
|
});
|
|
1527
1553
|
const arr = `[${fns.join(",")}]`;
|
|
1554
|
+
const safeEmbed = bundleUsesSafeRe ? SAFE_REGEX_EMBED + "\n" : "";
|
|
1528
1555
|
if (format === 'esm') {
|
|
1529
|
-
return `// Auto-generated by ata-validator — do not edit\
|
|
1556
|
+
return `// Auto-generated by ata-validator — do not edit\n${safeEmbed}const R=${R};\nconst validators=${arr};\nexport default validators;\nexport { validators };\n`;
|
|
1530
1557
|
}
|
|
1531
|
-
return `'use strict';\
|
|
1558
|
+
return `'use strict';\n${safeEmbed}var R=${R};\nmodule.exports=[${fns.join(",")}];\n`;
|
|
1532
1559
|
};
|
|
1533
1560
|
|
|
1534
1561
|
// Compact bundle: deduplicated code. Shared template functions + per-schema params.
|
|
@@ -1538,6 +1565,7 @@ Validator.bundleCompact = function (schemas, opts) {
|
|
|
1538
1565
|
const haveIds = schemas.some((s) => s && typeof s === 'object' && s.$id);
|
|
1539
1566
|
const bundleOpts = haveIds ? { ...(opts || {}), schemas } : (opts || {});
|
|
1540
1567
|
const format = (opts && opts.format) || 'cjs';
|
|
1568
|
+
let bundleUsesSafeRe = false;
|
|
1541
1569
|
// Analyze schemas and group by structure
|
|
1542
1570
|
const entries = schemas.map((schema) => {
|
|
1543
1571
|
const v = new Validator(schema, bundleOpts);
|
|
@@ -1548,6 +1576,7 @@ Validator.bundleCompact = function (schemas, opts) {
|
|
|
1548
1576
|
typeof schema === "string" ? JSON.parse(schema) : schema,
|
|
1549
1577
|
v._schemaMap,
|
|
1550
1578
|
);
|
|
1579
|
+
if (jsFn._usesSafeRe || (jsErrFn && jsErrFn._usesSafeRe)) bundleUsesSafeRe = true;
|
|
1551
1580
|
// Hoisted anyOf/oneOf branch helpers (e.g. `_af1_b0`) must travel with the
|
|
1552
1581
|
// hybrid body or it references undefined names. Prepending keeps dedup honest:
|
|
1553
1582
|
// schemas with different branch sets no longer collide on body alone.
|
|
@@ -1591,6 +1620,7 @@ Validator.bundleCompact = function (schemas, opts) {
|
|
|
1591
1620
|
let out = isEsm
|
|
1592
1621
|
? "// Auto-generated by ata-validator — do not edit\n"
|
|
1593
1622
|
: "'use strict';\n";
|
|
1623
|
+
if (bundleUsesSafeRe) out += SAFE_REGEX_EMBED + "\n";
|
|
1594
1624
|
const declKW = isEsm ? "const" : "var";
|
|
1595
1625
|
out += `${declKW} R=Object.freeze({valid:true,errors:Object.freeze([])});\n`;
|
|
1596
1626
|
|
package/lib/js-compiler.js
CHANGED
|
@@ -1,6 +1,15 @@
|
|
|
1
1
|
'use strict'
|
|
2
2
|
|
|
3
3
|
const { codeFor } = require('./error-codes')
|
|
4
|
+
const { compileSafe, patternIsSafe } = require('./safe-regex')
|
|
5
|
+
|
|
6
|
+
// Closure value for a patternProperties/propertyNames regex: the linear-time
|
|
7
|
+
// safe matcher when the pattern is in the supported subset (and flag the ctx so
|
|
8
|
+
// standalone embeds the engine), else a plain RegExp. Both expose `.test`.
|
|
9
|
+
function safeReClosure (ctx, src) {
|
|
10
|
+
if (patternIsSafe(src)) { ctx.usesSafeRe = true; return compileSafe(src) }
|
|
11
|
+
return new RegExp(src)
|
|
12
|
+
}
|
|
4
13
|
|
|
5
14
|
// Compile a JSON Schema into a pure JS validator function.
|
|
6
15
|
// Closure-based validator — no new Function() or eval().
|
|
@@ -328,7 +337,7 @@ function compileToJS(schema, defs, schemaMap) {
|
|
|
328
337
|
}
|
|
329
338
|
if (schema.pattern) {
|
|
330
339
|
try {
|
|
331
|
-
const re = new RegExp(schema.pattern)
|
|
340
|
+
const re = patternIsSafe(schema.pattern) ? compileSafe(schema.pattern) : new RegExp(schema.pattern)
|
|
332
341
|
checks.push((d) => typeof d !== 'string' || re.test(d))
|
|
333
342
|
} catch {
|
|
334
343
|
return null
|
|
@@ -962,6 +971,12 @@ function compileToJSCodegen(schema, schemaMap, userFormats) {
|
|
|
962
971
|
|
|
963
972
|
// Pre-create regex objects once
|
|
964
973
|
for (const code of ctx.helperCode) {
|
|
974
|
+
const safeMatch = code.match(/^const (_re\d+)=__ataSafeRe\((.+)\)$/)
|
|
975
|
+
if (safeMatch) {
|
|
976
|
+
closureNames.push(safeMatch[1])
|
|
977
|
+
closureValues.push(compileSafe(JSON.parse(safeMatch[2])))
|
|
978
|
+
continue
|
|
979
|
+
}
|
|
965
980
|
const match = code.match(/^const (_re\d+)=new RegExp\((.+)\)$/)
|
|
966
981
|
if (match) {
|
|
967
982
|
closureNames.push(match[1])
|
|
@@ -1005,6 +1020,7 @@ function compileToJSCodegen(schema, schemaMap, userFormats) {
|
|
|
1005
1020
|
boolFn._source = helperStr + body
|
|
1006
1021
|
boolFn._preambleSource = preambleStr
|
|
1007
1022
|
boolFn._hybridSource = helperStr + hybridBody
|
|
1023
|
+
boolFn._usesSafeRe = !!ctx.usesSafeRe
|
|
1008
1024
|
// Custom-format closure entries that the bundle output needs to recreate.
|
|
1009
1025
|
// Stored as { name, fn } so consumers can serialize via Function#toString.
|
|
1010
1026
|
if (ctx.userFormats) {
|
|
@@ -1454,7 +1470,12 @@ function genCode(schema, v, lines, ctx, knownType) {
|
|
|
1454
1470
|
if (!ctx.regExpMap.has(pattern)) {
|
|
1455
1471
|
const ri = ctx.varCounter++
|
|
1456
1472
|
ctx.regExpMap.set(pattern, ri)
|
|
1457
|
-
|
|
1473
|
+
if (patternIsSafe(schema.pattern)) {
|
|
1474
|
+
ctx.helperCode.push(`const _re${ri}=__ataSafeRe(${pattern})`);
|
|
1475
|
+
ctx.usesSafeRe = true
|
|
1476
|
+
} else {
|
|
1477
|
+
ctx.helperCode.push(`const _re${ri}=new RegExp(${pattern})`);
|
|
1478
|
+
}
|
|
1458
1479
|
}
|
|
1459
1480
|
const ri = ctx.regExpMap.get(pattern);
|
|
1460
1481
|
lines.push(isStr ? `if(!_re${ri}.test(${v}))return false` : `if(typeof ${v}==='string'&&!_re${ri}.test(${v}))return false`)
|
|
@@ -1552,7 +1573,7 @@ function genCode(schema, v, lines, ctx, knownType) {
|
|
|
1552
1573
|
} else {
|
|
1553
1574
|
const ri = ctx.varCounter++
|
|
1554
1575
|
ctx.closureVars.push(`_re${ri}`)
|
|
1555
|
-
ctx.closureVals.push(
|
|
1576
|
+
ctx.closureVals.push(safeReClosure(ctx, pat))
|
|
1556
1577
|
matchers.push({ check: `_re${ri}.test(${kVar})` })
|
|
1557
1578
|
}
|
|
1558
1579
|
}
|
|
@@ -1587,7 +1608,7 @@ function genCode(schema, v, lines, ctx, knownType) {
|
|
|
1587
1608
|
} else {
|
|
1588
1609
|
const ri = ctx.varCounter++
|
|
1589
1610
|
ctx.closureVars.push(`_re${ri}`)
|
|
1590
|
-
ctx.closureVals.push(
|
|
1611
|
+
ctx.closureVals.push(safeReClosure(ctx, pn.pattern))
|
|
1591
1612
|
lines.push(`if(!_re${ri}.test(${kVar}))return false`)
|
|
1592
1613
|
}
|
|
1593
1614
|
}
|
|
@@ -1630,7 +1651,7 @@ function genCode(schema, v, lines, ctx, knownType) {
|
|
|
1630
1651
|
} else {
|
|
1631
1652
|
const ri = ctx.varCounter++
|
|
1632
1653
|
ctx.closureVars.push(`_re${ri}`)
|
|
1633
|
-
ctx.closureVals.push(
|
|
1654
|
+
ctx.closureVals.push(safeReClosure(ctx, pn.pattern))
|
|
1634
1655
|
lines.push(`if(!_re${ri}.test(${kVar}))return false`)
|
|
1635
1656
|
}
|
|
1636
1657
|
}
|
|
@@ -1674,7 +1695,7 @@ function genCode(schema, v, lines, ctx, knownType) {
|
|
|
1674
1695
|
} else {
|
|
1675
1696
|
const ri = ctx.varCounter++
|
|
1676
1697
|
ctx.closureVars.push(`_re${ri}`)
|
|
1677
|
-
ctx.closureVals.push(
|
|
1698
|
+
ctx.closureVals.push(safeReClosure(ctx, pn.pattern))
|
|
1678
1699
|
lines.push(`if(!_re${ri}.test(_k${ki}))return false`)
|
|
1679
1700
|
}
|
|
1680
1701
|
}
|
|
@@ -2102,7 +2123,7 @@ function genCode(schema, v, lines, ctx, knownType) {
|
|
|
2102
2123
|
for (const pat of allPatterns) {
|
|
2103
2124
|
const ri = ctx.varCounter++
|
|
2104
2125
|
ctx.closureVars.push(`_ure${ri}`)
|
|
2105
|
-
ctx.closureVals.push(
|
|
2126
|
+
ctx.closureVals.push(safeReClosure(ctx, pat))
|
|
2106
2127
|
reVars.push(`_ure${ri}`)
|
|
2107
2128
|
}
|
|
2108
2129
|
if (schema.if && !schema.then && !schema.else) {
|
|
@@ -2119,7 +2140,7 @@ function genCode(schema, v, lines, ctx, knownType) {
|
|
|
2119
2140
|
for (const pat of ifPatterns) {
|
|
2120
2141
|
const ri = ctx.varCounter++
|
|
2121
2142
|
ctx.closureVars.push(`_ure${ri}`)
|
|
2122
|
-
ctx.closureVals.push(
|
|
2143
|
+
ctx.closureVals.push(safeReClosure(ctx, pat))
|
|
2123
2144
|
ifReVars.push(`_ure${ri}`)
|
|
2124
2145
|
}
|
|
2125
2146
|
const rootReVars = []
|
|
@@ -2127,7 +2148,7 @@ function genCode(schema, v, lines, ctx, knownType) {
|
|
|
2127
2148
|
for (const pat of Object.keys(schema.patternProperties)) {
|
|
2128
2149
|
const ri = ctx.varCounter++
|
|
2129
2150
|
ctx.closureVars.push(`_ure${ri}`)
|
|
2130
|
-
ctx.closureVals.push(
|
|
2151
|
+
ctx.closureVals.push(safeReClosure(ctx, pat))
|
|
2131
2152
|
rootReVars.push(`_ure${ri}`)
|
|
2132
2153
|
}
|
|
2133
2154
|
}
|
|
@@ -2621,8 +2642,17 @@ function compileToJSCodegenWithErrors(schema, schemaMap, userFormats, sourceOpts
|
|
|
2621
2642
|
`\n return{valid:_e.length===0,errors:_e}`
|
|
2622
2643
|
}
|
|
2623
2644
|
try {
|
|
2624
|
-
|
|
2645
|
+
let fn
|
|
2646
|
+
if (ctx.usesSafeRe) {
|
|
2647
|
+
// Inlined helperCode references __ataSafeRe; bind the safe-regex factory.
|
|
2648
|
+
// Standalone keeps the inline _errSource — the embed defines __ataSafeRe.
|
|
2649
|
+
const built = new Function('__ataSafeRe', 'd', '_all', body)
|
|
2650
|
+
fn = (d, _all) => built(compileSafe, d, _all)
|
|
2651
|
+
} else {
|
|
2652
|
+
fn = new Function('d', '_all', body)
|
|
2653
|
+
}
|
|
2625
2654
|
fn._errSource = body
|
|
2655
|
+
fn._usesSafeRe = !!ctx.usesSafeRe
|
|
2626
2656
|
return fn
|
|
2627
2657
|
} catch {
|
|
2628
2658
|
return null
|
|
@@ -2856,7 +2886,12 @@ function genCodeE(schema, v, pathExpr, lines, ctx, schemaPrefix) {
|
|
|
2856
2886
|
if (!ctx.regExpMap.has(pattern)) {
|
|
2857
2887
|
const ri = ctx.varCounter++
|
|
2858
2888
|
ctx.regExpMap.set(pattern, ri)
|
|
2859
|
-
|
|
2889
|
+
if (patternIsSafe(schema.pattern)) {
|
|
2890
|
+
ctx.helperCode.push(`const _re${ri}=__ataSafeRe(${pattern})`)
|
|
2891
|
+
ctx.usesSafeRe = true
|
|
2892
|
+
} else {
|
|
2893
|
+
ctx.helperCode.push(`const _re${ri}=new RegExp(${pattern})`)
|
|
2894
|
+
}
|
|
2860
2895
|
}
|
|
2861
2896
|
const ri = ctx.regExpMap.get(pattern);
|
|
2862
2897
|
const c = isStr ? `!_re${ri}.test(${v})` : `typeof ${v}==='string'&&!_re${ri}.test(${v})`
|
|
@@ -2961,7 +2996,12 @@ function genCodeE(schema, v, pathExpr, lines, ctx, schemaPrefix) {
|
|
|
2961
2996
|
if (!ctx.regExpMap.has(pattern)) {
|
|
2962
2997
|
const ri = ctx.varCounter++
|
|
2963
2998
|
ctx.regExpMap.set(pattern, ri)
|
|
2964
|
-
|
|
2999
|
+
if (patternIsSafe(pat)) {
|
|
3000
|
+
ctx.helperCode.push(`const _re${ri}=__ataSafeRe(${pattern})`);
|
|
3001
|
+
ctx.usesSafeRe = true
|
|
3002
|
+
} else {
|
|
3003
|
+
ctx.helperCode.push(`const _re${ri}=new RegExp(${pattern})`);
|
|
3004
|
+
}
|
|
2965
3005
|
}
|
|
2966
3006
|
const ri = ctx.regExpMap.get(pattern);
|
|
2967
3007
|
const ki = ctx.varCounter++
|
|
@@ -2997,7 +3037,12 @@ function genCodeE(schema, v, pathExpr, lines, ctx, schemaPrefix) {
|
|
|
2997
3037
|
if (!ctx.regExpMap.has(pattern)) {
|
|
2998
3038
|
const ri = ctx.varCounter++
|
|
2999
3039
|
ctx.regExpMap.set(pattern, ri)
|
|
3000
|
-
|
|
3040
|
+
if (patternIsSafe(pn.pattern)) {
|
|
3041
|
+
ctx.helperCode.push(`const _re${ri}=__ataSafeRe(${pattern})`);
|
|
3042
|
+
ctx.usesSafeRe = true
|
|
3043
|
+
} else {
|
|
3044
|
+
ctx.helperCode.push(`const _re${ri}=new RegExp(${pattern})`);
|
|
3045
|
+
}
|
|
3001
3046
|
}
|
|
3002
3047
|
const ri = ctx.regExpMap.get(pattern);
|
|
3003
3048
|
lines.push(`if(!_re${ri}.test(_k${ki})){${fail('pattern', 'propertyNames/pattern', `{pattern:${JSON.stringify(pn.pattern)}}`, `'must match pattern "${pn.pattern}"'`)}}`)
|
|
@@ -3419,7 +3464,7 @@ function genCodeC(schema, v, pathExpr, lines, ctx, schemaPrefix) {
|
|
|
3419
3464
|
const ri = ctx.varCounter++
|
|
3420
3465
|
const reVar = `_re${ri}`
|
|
3421
3466
|
ctx.closureVars.push(reVar)
|
|
3422
|
-
ctx.closureVals.push(new RegExp(schema.pattern))
|
|
3467
|
+
ctx.closureVals.push(patternIsSafe(schema.pattern) ? compileSafe(schema.pattern) : new RegExp(schema.pattern))
|
|
3423
3468
|
const c = isStr ? `!${reVar}.test(${v})` : `typeof ${v}==='string'&&!${reVar}.test(${v})`
|
|
3424
3469
|
lines.push(`if(${c}){${fail('pattern', 'pattern', `{pattern:${JSON.stringify(schema.pattern)}}`, `'must match pattern "${schema.pattern}"'`)}}`)
|
|
3425
3470
|
}
|
|
@@ -3544,7 +3589,7 @@ function genCodeC(schema, v, pathExpr, lines, ctx, schemaPrefix) {
|
|
|
3544
3589
|
} else {
|
|
3545
3590
|
const ri = ctx.varCounter++
|
|
3546
3591
|
ctx.closureVars.push(`_re${ri}`)
|
|
3547
|
-
ctx.closureVals.push(
|
|
3592
|
+
ctx.closureVals.push(safeReClosure(ctx, pat))
|
|
3548
3593
|
matchers.push({ check: `_re${ri}.test(_k${pi})` })
|
|
3549
3594
|
}
|
|
3550
3595
|
}
|
|
@@ -3588,7 +3633,7 @@ function genCodeC(schema, v, pathExpr, lines, ctx, schemaPrefix) {
|
|
|
3588
3633
|
} else {
|
|
3589
3634
|
const ri = ctx.varCounter++
|
|
3590
3635
|
ctx.closureVars.push(`_re${ri}`)
|
|
3591
|
-
ctx.closureVals.push(
|
|
3636
|
+
ctx.closureVals.push(safeReClosure(ctx, pn.pattern))
|
|
3592
3637
|
lines.push(`if(!_re${ri}.test(${kVar})){${fail('pattern', 'propertyNames/pattern', `{pattern:${JSON.stringify(pn.pattern)}}`, `'must match pattern "${pn.pattern}"'`)}}`)
|
|
3593
3638
|
}
|
|
3594
3639
|
}
|
|
@@ -3620,7 +3665,7 @@ function genCodeC(schema, v, pathExpr, lines, ctx, schemaPrefix) {
|
|
|
3620
3665
|
} else {
|
|
3621
3666
|
const ri = ctx.varCounter++
|
|
3622
3667
|
ctx.closureVars.push(`_re${ri}`)
|
|
3623
|
-
ctx.closureVals.push(
|
|
3668
|
+
ctx.closureVals.push(safeReClosure(ctx, pn.pattern))
|
|
3624
3669
|
lines.push(`if(!_re${ri}.test(${kVar})){${fail('pattern', 'propertyNames/pattern', `{pattern:${JSON.stringify(pn.pattern)}}`, `'must match pattern "${pn.pattern}"'`)}}`)
|
|
3625
3670
|
}
|
|
3626
3671
|
}
|
|
@@ -3662,7 +3707,7 @@ function genCodeC(schema, v, pathExpr, lines, ctx, schemaPrefix) {
|
|
|
3662
3707
|
if (pn.pattern) {
|
|
3663
3708
|
const ri = ctx.varCounter++
|
|
3664
3709
|
ctx.closureVars.push(`_re${ri}`)
|
|
3665
|
-
ctx.closureVals.push(
|
|
3710
|
+
ctx.closureVals.push(safeReClosure(ctx, pn.pattern))
|
|
3666
3711
|
lines.push(`if(!_re${ri}.test(_k${ki})){${fail('pattern', 'propertyNames/pattern', `{pattern:${JSON.stringify(pn.pattern)}}`, `'must match pattern "${pn.pattern}"'`)}}`)
|
|
3667
3712
|
}
|
|
3668
3713
|
if (pn.const !== undefined) {
|
|
@@ -0,0 +1,300 @@
|
|
|
1
|
+
'use strict'
|
|
2
|
+
|
|
3
|
+
// Linear-time regex engine for JSON Schema `pattern`, used in place of JS RegExp
|
|
4
|
+
// so an adversarial input cannot trigger catastrophic backtracking (ReDoS).
|
|
5
|
+
//
|
|
6
|
+
// It is a Pike VM: the pattern compiles to a small instruction program, and the
|
|
7
|
+
// VM simulates all NFA threads in lockstep over the input, deduping by program
|
|
8
|
+
// counter. Runtime is O(input * program), with no backtracking.
|
|
9
|
+
//
|
|
10
|
+
// Supported (the RE2 subset, which is what ata's native path also accepts):
|
|
11
|
+
// literals, ., character classes, \d \w \s \D \W \S, anchors ^ $, quantifiers
|
|
12
|
+
// * + ? {n} {n,} {n,m} (greedy or lazy, same language for a boolean test),
|
|
13
|
+
// groups ( ) (?: ), alternation |. Backreferences and lookaround are not
|
|
14
|
+
// supported by linear engines; compileSafe throws on them so the caller can
|
|
15
|
+
// decide (ata's codegen rejects such schemas rather than risk a hang).
|
|
16
|
+
|
|
17
|
+
const WS = [[9, 13], [32, 32], [160, 160]]
|
|
18
|
+
const DIGIT = [[48, 57]]
|
|
19
|
+
const WORD = [[48, 57], [65, 90], [97, 122], [95, 95]]
|
|
20
|
+
|
|
21
|
+
function parse (src) {
|
|
22
|
+
let i = 0
|
|
23
|
+
const len = src.length
|
|
24
|
+
const peek = () => src[i]
|
|
25
|
+
const eof = () => i >= len
|
|
26
|
+
|
|
27
|
+
function parseAlt () {
|
|
28
|
+
const opts = [parseConcat()]
|
|
29
|
+
while (!eof() && peek() === '|') { i++; opts.push(parseConcat()) }
|
|
30
|
+
return opts.length === 1 ? opts[0] : { t: 'alt', opts }
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
function parseConcat () {
|
|
34
|
+
const parts = []
|
|
35
|
+
while (!eof() && peek() !== '|' && peek() !== ')') parts.push(parseRepeat())
|
|
36
|
+
if (parts.length === 0) return { t: 'empty' }
|
|
37
|
+
return parts.length === 1 ? parts[0] : { t: 'concat', parts }
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
function parseRepeat () {
|
|
41
|
+
let node = parseAtom()
|
|
42
|
+
while (!eof()) {
|
|
43
|
+
const ch = peek()
|
|
44
|
+
if (ch === '*') { i++; node = { t: 'star', child: node } }
|
|
45
|
+
else if (ch === '+') { i++; node = { t: 'plus', child: node } }
|
|
46
|
+
else if (ch === '?') { i++; node = { t: 'quest', child: node } }
|
|
47
|
+
else if (ch === '{') {
|
|
48
|
+
const saved = i
|
|
49
|
+
const q = tryQuantifier()
|
|
50
|
+
if (!q) { i = saved; break }
|
|
51
|
+
node = { t: 'repeat', child: node, min: q.min, max: q.max }
|
|
52
|
+
} else break
|
|
53
|
+
// a trailing ? makes the quantifier lazy; same language for a boolean test
|
|
54
|
+
if (!eof() && peek() === '?') i++
|
|
55
|
+
}
|
|
56
|
+
return node
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
function tryQuantifier () {
|
|
60
|
+
// assumes current char is '{'
|
|
61
|
+
i++
|
|
62
|
+
let min = ''
|
|
63
|
+
while (!eof() && /[0-9]/.test(peek())) { min += peek(); i++ }
|
|
64
|
+
if (min === '') return null
|
|
65
|
+
let max
|
|
66
|
+
if (peek() === '}') { i++; return { min: +min, max: +min } }
|
|
67
|
+
if (peek() === ',') {
|
|
68
|
+
i++
|
|
69
|
+
let m = ''
|
|
70
|
+
while (!eof() && /[0-9]/.test(peek())) { m += peek(); i++ }
|
|
71
|
+
if (peek() !== '}') return null
|
|
72
|
+
i++
|
|
73
|
+
max = m === '' ? Infinity : +m
|
|
74
|
+
return { min: +min, max }
|
|
75
|
+
}
|
|
76
|
+
return null
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
function parseAtom () {
|
|
80
|
+
const ch = peek()
|
|
81
|
+
if (ch === '(') {
|
|
82
|
+
i++
|
|
83
|
+
if (src[i] === '?') {
|
|
84
|
+
if (src[i + 1] === ':') { i += 2 }
|
|
85
|
+
else throw new Error('unsupported group (lookaround/named) in pattern')
|
|
86
|
+
}
|
|
87
|
+
const child = parseAlt()
|
|
88
|
+
if (peek() !== ')') throw new Error('unbalanced ( in pattern')
|
|
89
|
+
i++
|
|
90
|
+
return { t: 'group', child }
|
|
91
|
+
}
|
|
92
|
+
if (ch === '[') return parseClass()
|
|
93
|
+
if (ch === '.') { i++; return { t: 'any' } }
|
|
94
|
+
if (ch === '^') { i++; return { t: 'bol' } }
|
|
95
|
+
if (ch === '$') { i++; return { t: 'eol' } }
|
|
96
|
+
if (ch === '\\') return parseEscape(false)
|
|
97
|
+
if (ch === ')' || ch === '|') return { t: 'empty' }
|
|
98
|
+
i++
|
|
99
|
+
return { t: 'char', c: ch.charCodeAt(0) }
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
function parseClass () {
|
|
103
|
+
i++ // [
|
|
104
|
+
let neg = false
|
|
105
|
+
if (peek() === '^') { neg = true; i++ }
|
|
106
|
+
const ranges = []
|
|
107
|
+
while (!eof() && peek() !== ']') {
|
|
108
|
+
let lo
|
|
109
|
+
if (peek() === '\\') {
|
|
110
|
+
const esc = parseEscape(true)
|
|
111
|
+
if (esc.t === 'classpart') { for (const r of esc.ranges) ranges.push(r); continue }
|
|
112
|
+
lo = esc.c
|
|
113
|
+
} else { lo = peek().charCodeAt(0); i++ }
|
|
114
|
+
if (peek() === '-' && src[i + 1] !== ']' && i + 1 < len) {
|
|
115
|
+
i++ // -
|
|
116
|
+
let hi
|
|
117
|
+
if (peek() === '\\') { const e = parseEscape(true); hi = e.c } else { hi = peek().charCodeAt(0); i++ }
|
|
118
|
+
ranges.push([lo, hi])
|
|
119
|
+
} else {
|
|
120
|
+
ranges.push([lo, lo])
|
|
121
|
+
}
|
|
122
|
+
}
|
|
123
|
+
if (peek() !== ']') throw new Error('unbalanced [ in pattern')
|
|
124
|
+
i++
|
|
125
|
+
return { t: 'class', neg, ranges }
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
function parseEscape (inClass) {
|
|
129
|
+
i++ // backslash
|
|
130
|
+
if (eof()) throw new Error('trailing backslash in pattern')
|
|
131
|
+
const ch = peek(); i++
|
|
132
|
+
switch (ch) {
|
|
133
|
+
case 'd': return inClass ? { t: 'classpart', ranges: DIGIT } : { t: 'class', neg: false, ranges: DIGIT }
|
|
134
|
+
case 'w': return inClass ? { t: 'classpart', ranges: WORD } : { t: 'class', neg: false, ranges: WORD }
|
|
135
|
+
case 's': return inClass ? { t: 'classpart', ranges: WS } : { t: 'class', neg: false, ranges: WS }
|
|
136
|
+
case 'D': if (inClass) throw new Error('\\D inside a class is not supported'); return { t: 'class', neg: true, ranges: DIGIT }
|
|
137
|
+
case 'W': if (inClass) throw new Error('\\W inside a class is not supported'); return { t: 'class', neg: true, ranges: WORD }
|
|
138
|
+
case 'S': if (inClass) throw new Error('\\S inside a class is not supported'); return { t: 'class', neg: true, ranges: WS }
|
|
139
|
+
case 'n': return { t: 'char', c: 10 }
|
|
140
|
+
case 'r': return { t: 'char', c: 13 }
|
|
141
|
+
case 't': return { t: 'char', c: 9 }
|
|
142
|
+
case 'f': return { t: 'char', c: 12 }
|
|
143
|
+
case 'v': return { t: 'char', c: 11 }
|
|
144
|
+
case '0': return { t: 'char', c: 0 }
|
|
145
|
+
case 'x': { const h = src.slice(i, i + 2); i += 2; return { t: 'char', c: parseInt(h, 16) } }
|
|
146
|
+
case 'u': { const h = src.slice(i, i + 4); i += 4; return { t: 'char', c: parseInt(h, 16) } }
|
|
147
|
+
case 'b': if (inClass) return { t: 'char', c: 8 }; throw new Error('\\b word boundary is not supported')
|
|
148
|
+
default:
|
|
149
|
+
if (/[1-9]/.test(ch)) throw new Error('backreferences are not supported in pattern')
|
|
150
|
+
return { t: 'char', c: ch.charCodeAt(0) }
|
|
151
|
+
}
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
const ast = parseAlt()
|
|
155
|
+
if (!eof()) throw new Error('unexpected "' + peek() + '" in pattern')
|
|
156
|
+
return ast
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
function compileProg (ast) {
|
|
160
|
+
const prog = []
|
|
161
|
+
const emit = (op, extra) => { const idx = prog.length; prog.push(Object.assign({ op }, extra)); return idx }
|
|
162
|
+
|
|
163
|
+
function rec (n) {
|
|
164
|
+
switch (n.t) {
|
|
165
|
+
case 'empty': break
|
|
166
|
+
case 'char': emit('char', { c: n.c }); break
|
|
167
|
+
case 'any': emit('any'); break
|
|
168
|
+
case 'class': emit('class', { neg: n.neg, ranges: n.ranges }); break
|
|
169
|
+
case 'bol': emit('bol'); break
|
|
170
|
+
case 'eol': emit('eol'); break
|
|
171
|
+
case 'group': rec(n.child); break
|
|
172
|
+
case 'concat': for (const p of n.parts) rec(p); break
|
|
173
|
+
case 'alt': {
|
|
174
|
+
const jmps = []
|
|
175
|
+
for (let k = 0; k < n.opts.length; k++) {
|
|
176
|
+
if (k < n.opts.length - 1) {
|
|
177
|
+
const sp = emit('split', { x: 0, y: 0 })
|
|
178
|
+
prog[sp].x = prog.length
|
|
179
|
+
rec(n.opts[k])
|
|
180
|
+
jmps.push(emit('jmp', { x: 0 }))
|
|
181
|
+
prog[sp].y = prog.length
|
|
182
|
+
} else {
|
|
183
|
+
rec(n.opts[k])
|
|
184
|
+
}
|
|
185
|
+
}
|
|
186
|
+
for (const j of jmps) prog[j].x = prog.length
|
|
187
|
+
break
|
|
188
|
+
}
|
|
189
|
+
case 'star': {
|
|
190
|
+
const sp = emit('split', { x: 0, y: 0 })
|
|
191
|
+
prog[sp].x = prog.length
|
|
192
|
+
rec(n.child)
|
|
193
|
+
emit('jmp', { x: sp })
|
|
194
|
+
prog[sp].y = prog.length
|
|
195
|
+
break
|
|
196
|
+
}
|
|
197
|
+
case 'plus': {
|
|
198
|
+
const start = prog.length
|
|
199
|
+
rec(n.child)
|
|
200
|
+
const sp = emit('split', { x: start, y: 0 })
|
|
201
|
+
prog[sp].y = prog.length
|
|
202
|
+
break
|
|
203
|
+
}
|
|
204
|
+
case 'quest': {
|
|
205
|
+
const sp = emit('split', { x: 0, y: 0 })
|
|
206
|
+
prog[sp].x = prog.length
|
|
207
|
+
rec(n.child)
|
|
208
|
+
prog[sp].y = prog.length
|
|
209
|
+
break
|
|
210
|
+
}
|
|
211
|
+
case 'repeat': {
|
|
212
|
+
for (let k = 0; k < n.min; k++) rec(n.child)
|
|
213
|
+
if (n.max === Infinity) {
|
|
214
|
+
if (n.min === 0) rec({ t: 'star', child: n.child })
|
|
215
|
+
else rec({ t: 'star', child: n.child })
|
|
216
|
+
} else {
|
|
217
|
+
for (let k = 0; k < n.max - n.min; k++) rec({ t: 'quest', child: n.child })
|
|
218
|
+
}
|
|
219
|
+
break
|
|
220
|
+
}
|
|
221
|
+
}
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
rec(ast)
|
|
225
|
+
emit('match')
|
|
226
|
+
return prog
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
function matchClass (instr, c) {
|
|
230
|
+
let inside = false
|
|
231
|
+
const r = instr.ranges
|
|
232
|
+
for (let k = 0; k < r.length; k++) { if (c >= r[k][0] && c <= r[k][1]) { inside = true; break } }
|
|
233
|
+
return instr.neg ? !inside : inside
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
function makeRunner (prog) {
|
|
237
|
+
const n = prog.length
|
|
238
|
+
const lastGen = new Int32Array(n).fill(-1)
|
|
239
|
+
let gen = 0
|
|
240
|
+
const stack = []
|
|
241
|
+
|
|
242
|
+
function addThread (list, pc, pos, len) {
|
|
243
|
+
stack.length = 0
|
|
244
|
+
stack.push(pc)
|
|
245
|
+
while (stack.length) {
|
|
246
|
+
const p = stack.pop()
|
|
247
|
+
if (lastGen[p] === gen) continue
|
|
248
|
+
lastGen[p] = gen
|
|
249
|
+
const I = prog[p]
|
|
250
|
+
switch (I.op) {
|
|
251
|
+
case 'jmp': stack.push(I.x); break
|
|
252
|
+
case 'split': stack.push(I.y); stack.push(I.x); break
|
|
253
|
+
case 'bol': if (pos === 0) stack.push(p + 1); break
|
|
254
|
+
case 'eol': if (pos === len) stack.push(p + 1); break
|
|
255
|
+
default: list.push(p)
|
|
256
|
+
}
|
|
257
|
+
}
|
|
258
|
+
}
|
|
259
|
+
|
|
260
|
+
return function test (s) {
|
|
261
|
+
const len = s.length
|
|
262
|
+
let clist = []
|
|
263
|
+
let nlist = []
|
|
264
|
+
gen++
|
|
265
|
+
addThread(clist, 0, 0, len)
|
|
266
|
+
for (let pos = 0; pos <= len; pos++) {
|
|
267
|
+
const c = pos < len ? s.charCodeAt(pos) : -1
|
|
268
|
+
gen++
|
|
269
|
+
nlist.length = 0
|
|
270
|
+
for (let k = 0; k < clist.length; k++) {
|
|
271
|
+
const pc = clist[k]
|
|
272
|
+
const I = prog[pc]
|
|
273
|
+
if (I.op === 'match') return true
|
|
274
|
+
else if (I.op === 'char') { if (c === I.c) addThread(nlist, pc + 1, pos + 1, len) }
|
|
275
|
+
else if (I.op === 'any') { if (c !== -1 && c !== 10) addThread(nlist, pc + 1, pos + 1, len) }
|
|
276
|
+
else if (I.op === 'class') { if (c !== -1 && matchClass(I, c)) addThread(nlist, pc + 1, pos + 1, len) }
|
|
277
|
+
}
|
|
278
|
+
if (pos < len) addThread(nlist, 0, pos + 1, len)
|
|
279
|
+
const tmp = clist; clist = nlist; nlist = tmp
|
|
280
|
+
}
|
|
281
|
+
return false
|
|
282
|
+
}
|
|
283
|
+
}
|
|
284
|
+
|
|
285
|
+
function compileSafe (pattern) {
|
|
286
|
+
const prog = compileProg(parse(pattern))
|
|
287
|
+
const runner = makeRunner(prog)
|
|
288
|
+
// `__ataSafe` brands the result so the standalone serializer can tell a safe
|
|
289
|
+
// matcher apart from a RegExp and emit `__ataSafeRe(source)` instead.
|
|
290
|
+
return { test: runner, source: pattern, __ataSafe: true }
|
|
291
|
+
}
|
|
292
|
+
|
|
293
|
+
// True when the linear engine can represent `src`. Used by the codegen to decide
|
|
294
|
+
// between the safe matcher and a JS RegExp fallback for patterns outside the
|
|
295
|
+
// supported (RE2) subset (backreferences, lookaround, etc.).
|
|
296
|
+
function patternIsSafe (src) {
|
|
297
|
+
try { compileSafe(src); return true } catch { return false }
|
|
298
|
+
}
|
|
299
|
+
|
|
300
|
+
module.exports = { compileSafe, patternIsSafe }
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "ata-validator",
|
|
3
|
-
"version": "0.17.
|
|
3
|
+
"version": "0.17.3",
|
|
4
4
|
"description": "JSON Schema validation with first-class TypeScript and zero runtime cost. AOT compile to per-schema ESM modules with zero validator dependency. Generic Validator<T> for TypeBox/Zod/Valibot composition. Optional runtime API. Standard Schema V1 compatible.",
|
|
5
5
|
"main": "index.js",
|
|
6
6
|
"module": "index.mjs",
|
|
@@ -42,7 +42,7 @@
|
|
|
42
42
|
"rebuild": "cmake-js rebuild --target ata",
|
|
43
43
|
"prebuild": "pkg-prebuilds-copy --baseDir build/Release --source ata.node --name=ata --strip --napi_version=10",
|
|
44
44
|
"prebuild-all": "npm run prebuild -- --arch x64 && npm run prebuild -- --arch arm64",
|
|
45
|
-
"test": "node test.js && node tests/test_no_native.js && node tests/test_aot_build.js && node tests/test_aot_differential.js && node tests/test_aot_cli_build.js && node tests/test_aot_cli_smoke.js && node tests/test_bundle_standalone.js && node tests/test_standalone_anyof.js && node tests/test_typed_validator_runner.js && node tests/test_define_schema.js && node tests/test_error_codes_lock.js && node tests/test_nullable.js && node tests/test_validate_and_parse.js && node tests/test_validate_data.js && node tests/test_enrich_error.js && node tests/test_rich_errors_optout.js && node tests/test_source_positions.js && node tests/fuzz_positions.js && node tests/test_data_positions.js && node tests/test_render_shared.js && node tests/test_renderers.js && node tests/test_runtime_error_dx.js && node tests/test_aot_error_dx.js && node tests/test_abort_early.js && node tests/test_branch_collapse.js && node tests/test_suggestions.js && node tests/test_cli_validate.js && node benchmark/bench_aot_size.mjs",
|
|
45
|
+
"test": "node test.js && node tests/test_no_native.js && node tests/test_safe_regex.js && node tests/test_safe_regex_integration.js && node tests/test_aot_build.js && node tests/test_aot_differential.js && node tests/test_aot_cli_build.js && node tests/test_aot_cli_smoke.js && node tests/test_bundle_standalone.js && node tests/test_standalone_anyof.js && node tests/test_typed_validator_runner.js && node tests/test_define_schema.js && node tests/test_error_codes_lock.js && node tests/test_nullable.js && node tests/test_validate_and_parse.js && node tests/test_validate_data.js && node tests/test_enrich_error.js && node tests/test_rich_errors_optout.js && node tests/test_source_positions.js && node tests/fuzz_positions.js && node tests/test_data_positions.js && node tests/test_render_shared.js && node tests/test_renderers.js && node tests/test_runtime_error_dx.js && node tests/test_aot_error_dx.js && node tests/test_abort_early.js && node tests/test_branch_collapse.js && node tests/test_suggestions.js && node tests/test_cli_validate.js && node benchmark/bench_aot_size.mjs",
|
|
46
46
|
"bench:size": "node benchmark/bench_aot_size.mjs",
|
|
47
47
|
"test:suite": "node tests/run_suite.js",
|
|
48
48
|
"test:compat": "node tests/test_compat.js",
|