ata-validator 1.36.1 → 1.37.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/build.d.ts CHANGED
@@ -178,3 +178,10 @@ export function compiledSchemaFor(schema: unknown): object;
178
178
  * whose detailed errors the generator cannot produce.
179
179
  */
180
180
  export function compiledModuleFor(schema: unknown, opts?: { format?: 'esm' | 'cjs' }): string | null;
181
+
182
+ /**
183
+ * The Validator options `fromCompiled()` reproduces, so a plugin can tell
184
+ * whether `new Validator(schema, options)` can be replaced. Absent before
185
+ * ata-validator 1.37.0, where only calls without options can be.
186
+ */
187
+ export const compiledOptions: readonly string[];
package/build.mjs CHANGED
@@ -12,4 +12,5 @@ export const schemaHash = mod.schemaHash;
12
12
  export const compiledEligible = mod.compiledEligible;
13
13
  export const compiledSchemaFor = mod.compiledSchemaFor;
14
14
  export const compiledModuleFor = mod.compiledModuleFor;
15
+ export const compiledOptions = mod.compiledOptions;
15
16
  export default mod;
package/compiled.d.ts CHANGED
@@ -14,4 +14,12 @@ export interface CompiledValidator<T = unknown> {
14
14
  isValidJSON(json: string): boolean;
15
15
  }
16
16
 
17
- export function fromCompiled<T = unknown>(mod: CompiledModule, schema: object): CompiledValidator<T>;
17
+ /** The Validator options `fromCompiled` reproduces. Any other option throws. */
18
+ export const COMPILED_OPTIONS: readonly ['useDefaults'];
19
+
20
+ export interface CompiledOptions {
21
+ /** `false` leaves the input unchanged, as `new Validator(schema, { useDefaults: false })` does. */
22
+ useDefaults?: boolean;
23
+ }
24
+
25
+ export function fromCompiled<T = unknown>(mod: CompiledModule, schema: object, options?: CompiledOptions): CompiledValidator<T>;
package/compiled.mjs CHANGED
@@ -1,3 +1,3 @@
1
1
  import mod from './compiled.js';
2
- export const { fromCompiled } = mod;
2
+ export const { fromCompiled, COMPILED_OPTIONS } = mod;
3
3
  export default mod;
package/index.d.ts CHANGED
@@ -452,6 +452,17 @@ export interface BundleStandaloneOptions extends ValidatorOptions {
452
452
  export interface StandardSchemaV1Props<Output = unknown, Input = unknown> {
453
453
  version: 1;
454
454
  vendor: "ata-validator";
455
+ /**
456
+ * Standard JSON Schema: the schema this validator checks, as a copy, for a
457
+ * consumer that publishes it (the MCP SDK's `inputSchema`, for one). The
458
+ * target must be the dialect the schema declares, and a schema without
459
+ * `$schema` is 2020-12. Any other target throws; ata does not convert
460
+ * between dialects.
461
+ */
462
+ jsonSchema: {
463
+ input(options: { target: "draft-2020-12" | "draft-07" | (string & {}); libraryOptions?: Record<string, unknown> }): Record<string, unknown>;
464
+ output(options: { target: "draft-2020-12" | "draft-07" | (string & {}); libraryOptions?: Record<string, unknown> }): Record<string, unknown>;
465
+ };
455
466
  validate(
456
467
  value: unknown
457
468
  ):
package/lib/aot-build.js CHANGED
@@ -295,8 +295,8 @@ const { schemaHash } = require('./schema-hash');
295
295
  // tests/test_compiled_parity.js holds that. Declined: custom error messages,
296
296
  // which the core applies in a layer the wrapper does not carry. A caller must also get a
297
297
  // module back from toStandaloneModule, which returns null for a schema it
298
- // cannot compile. The text checks are deliberately coarse: a property named
299
- // `default` declines too, and declining costs only the saving.
298
+ // cannot compile. The text check is deliberately coarse: a property named
299
+ // `errorMessage` declines too, and declining costs only the saving.
300
300
  function compiledEligible(schema) {
301
301
  if (typeof schema !== 'object' || schema === null || Array.isArray(schema)) return false;
302
302
  return !JSON.stringify(schema).includes('"errorMessage"');
@@ -331,7 +331,13 @@ function compiledModuleFor(schema, opts) {
331
331
  return src && !degraded ? src : null;
332
332
  }
333
333
 
334
+ // The Validator options fromCompiled() reproduces, for a plugin deciding whether
335
+ // `new Validator(schema, options)` can be replaced. An older ata-validator does
336
+ // not export this, and a plugin should then replace only calls without options.
337
+ const { COMPILED_OPTIONS: compiledOptions } = require('./compiled');
338
+
334
339
  module.exports = {
340
+ compiledOptions,
335
341
  compiledEligible,
336
342
  compiledSchemaFor,
337
343
  compiledModuleFor,
@@ -37,6 +37,30 @@ const SUBSCHEMA_NODES = [
37
37
  'allOf', 'anyOf', 'oneOf', 'prefixItems',
38
38
  ];
39
39
 
40
+ // Where RE2 and ECMA-262 give different answers for the same pattern. RE2's
41
+ // `\s` is the ASCII whitespace alone, so it misses U+00A0, U+2028, U+3000 and
42
+ // U+FEFF, which ECMA-262 counts, and its `.` matches a carriage return and the
43
+ // line and paragraph separators, which ECMA-262's does not. `^\S+$` accepted
44
+ // "a\u2028b" on the native text path while validate() rejected it. Any `\s`
45
+ // or `\S`, and any `.` outside a class that is not escaped, sends the schema to
46
+ // the JS path.
47
+ function re2Diverges(pattern) {
48
+ let inClass = false;
49
+ for (let i = 0; i < pattern.length; i++) {
50
+ const c = pattern[i];
51
+ if (c === '\\') {
52
+ const e = pattern[i + 1];
53
+ if (e === 's' || e === 'S') return true;
54
+ i++;
55
+ continue;
56
+ }
57
+ if (inClass) { if (c === ']') inClass = false; continue; }
58
+ if (c === '[') { inClass = true; continue; }
59
+ if (c === '.') return true;
60
+ }
61
+ return false;
62
+ }
63
+
40
64
  function walk(schema, depth) {
41
65
  if (schema === true || schema === false) {
42
66
  // A boolean root is answered wrong by the walker; nested booleans are
@@ -62,6 +86,7 @@ function walk(schema, depth) {
62
86
  // Unicode property escapes: RE2 cannot parse them and the walker then
63
87
  // skips the pattern instead of failing.
64
88
  if (key === 'pattern' && typeof v === 'string' && /\\[pP]\{/.test(v)) return true;
89
+ if (key === 'pattern' && typeof v === 'string' && re2Diverges(v)) return true;
65
90
  // Tuple forms: prefixItems and the draft-07 array form of items are
66
91
  // checked against the wrong positions by the walker.
67
92
  if (key === 'prefixItems') return true;
package/lib/compiled.js CHANGED
@@ -10,14 +10,20 @@
10
10
  //
11
11
  // Only for a schema compiledEligible() accepts, and with the schema
12
12
  // compiledSchemaFor() returns: the one the runtime reads, after normalization.
13
- // Custom error messages and every option other than the defaults stay on the
14
- // runtime. Defaults are applied as a default Validator applies them, before
15
- // any check, with the same code.
13
+ // Custom error messages stay on the runtime, and so does every option except
14
+ // the ones COMPILED_OPTIONS lists. Defaults are applied as a default Validator
15
+ // applies them, before any check, with the same code; `useDefaults: false`
16
+ // leaves the input as it is, as the runtime then does.
16
17
 
17
18
  const { LazyRejection, RichRejection, LazyJsonRejection, _enrichLazy } = require('./rejections');
18
19
  const { buildDefaultsApplier } = require('./defaults');
19
20
 
20
21
  const VALID_RESULT = Object.freeze({ valid: true, errors: Object.freeze([]) });
22
+
23
+ // The Validator options the wrapper answers the same way as the runtime.
24
+ // Anything else changes what the runtime does in a way the wrapper does not
25
+ // reproduce, so it is refused rather than ignored.
26
+ const COMPILED_OPTIONS = Object.freeze(['useDefaults']);
21
27
  const EMPTY_ERRORS = Object.freeze([]);
22
28
  const VERDICT_DISAGREES = Object.freeze({
23
29
  valid: false,
@@ -43,9 +49,17 @@ class CompiledState {
43
49
  }
44
50
  }
45
51
 
46
- function fromCompiled(mod, schema) {
52
+ function fromCompiled(mod, schema, options) {
53
+ if (options !== undefined && (options === null || typeof options !== 'object')) {
54
+ throw new TypeError('fromCompiled options must be an object');
55
+ }
56
+ for (const key of Object.keys(options || {})) {
57
+ if (!COMPILED_OPTIONS.includes(key)) {
58
+ throw new TypeError(`fromCompiled does not support the ${key} option; a Validator with it has to stay on the runtime`);
59
+ }
60
+ }
47
61
  const self = new CompiledState(schema);
48
- const fill = buildDefaultsApplier(schema);
62
+ const fill = options && options.useDefaults === false ? null : buildDefaultsApplier(schema);
49
63
  if (fill) {
50
64
  self._mutatesInput = true;
51
65
  self._preprocess = fill;
@@ -93,4 +107,4 @@ function fromCompiled(mod, schema) {
93
107
  };
94
108
  }
95
109
 
96
- module.exports = { fromCompiled };
110
+ module.exports = { fromCompiled, COMPILED_OPTIONS };
@@ -1237,6 +1237,30 @@ const REF_NEUTRAL_SIBLINGS = new Set([
1237
1237
  'markdownEnumDescriptions', 'defaultSnippets', 'doNotSuggest', 'suggestSortText',
1238
1238
  ])
1239
1239
 
1240
+ // Stands for the schema path of a call to a recursive definition's error
1241
+ // helper while its body is generated; see genCodeENode. Not a pointer, so no
1242
+ // ordinal or source frame is looked up for it at compile time.
1243
+ const SP_PLACEHOLDER = '@@ata-sp@@'
1244
+
1245
+ // Follows a definition that is a local $ref to another definition until it
1246
+ // reaches one that is not. False for a chain that loops, and for one whose
1247
+ // next hop is not a definition in this document; codegenSafe decides whether
1248
+ // a non-local reference resolves.
1249
+ function aliasChainEnds(def, defs) {
1250
+ const seen = new Set()
1251
+ let node = def
1252
+ while (node && typeof node === 'object' && typeof node.$ref === 'string') {
1253
+ const m = node.$ref.match(/^#\/(?:\$defs|definitions)\/([^/]+)$/)
1254
+ if (!m) return !node.$ref.startsWith('#/')
1255
+ const name = m[1].replace(/~1/g, '/').replace(/~0/g, '~')
1256
+ if (seen.has(name)) return false
1257
+ seen.add(name)
1258
+ node = defs[name]
1259
+ if (node === undefined) return false
1260
+ }
1261
+ return true
1262
+ }
1263
+
1240
1264
  function codegenSafe(schema, schemaMap) {
1241
1265
  if (typeof schema === 'boolean') return true
1242
1266
  if (typeof schema !== 'object' || schema === null) return true
@@ -1381,7 +1405,11 @@ function codegenSafe(schema, schemaMap) {
1381
1405
  // codegen anchor maps register it (see "Build anchors map"), so anchor
1382
1406
  // refs (`$ref: '#name'`) resolve to it on all paths.
1383
1407
  if (def.$id && !def.$id.startsWith('#')) return false
1384
- if (def.$ref) return false // nested ref chain — bail
1408
+ // A definition that is only a reference to another one is an alias,
1409
+ // the shape TypeScript-to-schema generators emit for a named union.
1410
+ // The generators resolve it like any other $ref; a chain of aliases
1411
+ // that loops back on itself has no schema at its end and declines.
1412
+ if (def.$ref && !aliasChainEnds(def, defs)) return false
1385
1413
  if (!codegenSafe(def, schemaMap)) return false
1386
1414
  }
1387
1415
  }
@@ -1697,7 +1725,103 @@ function sharedCodegenGate(schema, schemaMap) {
1697
1725
 
1698
1726
  // --- Codegen mode: generates a single Function (NOT CSP-safe) ---
1699
1727
  // This matches ajv's approach: one monolithic function, V8 JIT fully inlines it
1728
+ // `propertyNames` as the generators read it. They express maxLength,
1729
+ // minLength, pattern, const and enum, and the gate declines anything else, so
1730
+ // a schema that writes `propertyNames: { $ref: '#/definitions/Key' }` or adds
1731
+ // `type: 'string'` went to the interpreted engine. Both are rewritten here, on
1732
+ // a copy of the nodes that change: a local reference, through aliases, is
1733
+ // replaced by what it names, `type: 'string'` is dropped since every property
1734
+ // name is a string, and annotations go. Where what is left is still not all
1735
+ // supported keys, the node is left as written and the gate declines as before.
1736
+ // Errors keep the path the interpreted engine reports, which runs through the
1737
+ // reference without naming it.
1738
+ const PN_SUPPORTED = new Set(['maxLength', 'minLength', 'pattern', 'const', 'enum'])
1739
+ const PN_ANNOTATIONS = new Set(['$schema', '$comment', 'title', 'description', 'examples', 'default', 'deprecated', 'readOnly', 'writeOnly'])
1740
+ function simplePropertyNames(pn, defs) {
1741
+ const seen = new Set()
1742
+ while (pn && typeof pn === 'object' && typeof pn.$ref === 'string') {
1743
+ if (Object.keys(pn).some((k) => k !== '$ref' && !PN_ANNOTATIONS.has(k) && !k.startsWith('x-'))) return null
1744
+ const m = pn.$ref.match(/^#\/(?:\$defs|definitions)\/([^/]+)$/)
1745
+ if (!m || !defs) return null
1746
+ const name = m[1].replace(/~1/g, '/').replace(/~0/g, '~')
1747
+ if (seen.has(name)) return null
1748
+ seen.add(name)
1749
+ pn = defs[name]
1750
+ }
1751
+ if (!pn || typeof pn !== 'object' || Array.isArray(pn)) return null
1752
+ const out = {}
1753
+ for (const [k, val] of Object.entries(pn)) {
1754
+ if (PN_ANNOTATIONS.has(k) || k.startsWith('x-')) continue
1755
+ if (k === 'type') {
1756
+ if (val === 'string' || (Array.isArray(val) && val.includes('string'))) continue
1757
+ return null
1758
+ }
1759
+ if (!PN_SUPPORTED.has(k)) return null
1760
+ out[k] = val
1761
+ }
1762
+ return out
1763
+ }
1764
+ // The three generators are handed the same schema object for one validator,
1765
+ // so the rewrite is kept per object and done once. It walks the schema rather
1766
+ // than serializing it to look for the keyword: on a 7.6 MB schema the
1767
+ // serialization, three times over, was 3% of the first answer.
1768
+ const _pnNormalized = new WeakMap()
1769
+ function normalizePropertyNames(root) {
1770
+ if (!root || typeof root !== 'object') return root
1771
+ const hit = _pnNormalized.get(root)
1772
+ if (hit !== undefined) return hit
1773
+ const out = normalizePropertyNamesWalk(root)
1774
+ _pnNormalized.set(root, out)
1775
+ return out
1776
+ }
1777
+ function normalizePropertyNamesWalk(root) {
1778
+ const defs = root.$defs || root.definitions || null
1779
+ const seen = new Map()
1780
+ const walk = (node) => {
1781
+ if (node === null || typeof node !== 'object') return node
1782
+ if (seen.has(node)) return seen.get(node)
1783
+ seen.set(node, node)
1784
+ if (Array.isArray(node)) {
1785
+ let out = node
1786
+ node.forEach((n, i) => { const w = walk(n); if (w !== n) { if (out === node) out = node.slice(); out[i] = w } })
1787
+ seen.set(node, out)
1788
+ return out
1789
+ }
1790
+ let out = node
1791
+ const set = (k, v) => { if (out === node) out = { ...node }; out[k] = v }
1792
+ for (const k of SUBSCHEMA_MAPS) {
1793
+ const map = node[k]
1794
+ if (!map || typeof map !== 'object') continue
1795
+ let copy = map
1796
+ for (const [name, sub] of Object.entries(map)) {
1797
+ const w = walk(sub)
1798
+ if (w !== sub) { if (copy === map) copy = { ...map }; copy[name] = w }
1799
+ }
1800
+ if (copy !== map) set(k, copy)
1801
+ }
1802
+ for (const k of SUBSCHEMA_LISTS) {
1803
+ if (!Array.isArray(node[k])) continue
1804
+ const w = walk(node[k])
1805
+ if (w !== node[k]) set(k, w)
1806
+ }
1807
+ for (const k of SUBSCHEMA_SINGLES) {
1808
+ if (k === 'propertyNames' || node[k] === null || typeof node[k] !== 'object') continue
1809
+ const w = walk(node[k])
1810
+ if (w !== node[k]) set(k, w)
1811
+ }
1812
+ if (node.propertyNames && typeof node.propertyNames === 'object') {
1813
+ const simple = simplePropertyNames(node.propertyNames, defs)
1814
+ if (simple && JSON.stringify(simple) !== JSON.stringify(node.propertyNames)) set('propertyNames', simple)
1815
+ }
1816
+ seen.set(node, out)
1817
+ return out
1818
+ }
1819
+ return walk(root)
1820
+ }
1821
+
1700
1822
  function compileToJSCodegen(schema, schemaMap, userFormats, opts) {
1823
+ const inputSchema = schema
1824
+ schema = normalizePropertyNames(schema)
1701
1825
  if (typeof schema === 'boolean') return schema ? () => true : () => false
1702
1826
  if (typeof schema !== 'object' || schema === null) return null
1703
1827
  // removeAdditional: a verdict function that deletes the keys the remover
@@ -1768,7 +1892,7 @@ function compileToJSCodegen(schema, schemaMap, userFormats, opts) {
1768
1892
  if (hasUnresolvableRef(schema, rootDefs, anchors, schemaMap, new Set())) return null
1769
1893
  if (needsBaseTracking(schema, schemaMap, new Set())) return null
1770
1894
 
1771
- const ctx = { varCounter: 0, helpers: [], helperCode: [], preamble: [], shared: [], closureVars: ['_cpLen'], closureVals: [_cpLen], rootDefs, refStack: new Set(), schemaMap: schemaMap || null, anchors, rootSchema: schema, userFormats: userFormats || null, removeNodes }
1895
+ const ctx = { varCounter: 0, helpers: [], helperCode: [], preamble: [], shared: [], closureVars: ['_cpLen'], closureVals: [_cpLen], rootDefs, refStack: new Set(), schemaMap: schemaMap || null, anchors, rootSchema: inputSchema, userFormats: userFormats || null, removeNodes }
1772
1896
  const lines = []
1773
1897
  try {
1774
1898
  genCode(schema, 'd', lines, ctx)
@@ -3912,6 +4036,8 @@ function genCharCodeSwitch(keys, v) {
3912
4036
  // Returns a function: (data, allErrors) => { valid, errors }
3913
4037
  // Valid path is still fast — only error path does extra work.
3914
4038
  function compileToJSCodegenWithErrors(schema, schemaMap, userFormats, sourceOpts) {
4039
+ const inputSchema = schema
4040
+ schema = normalizePropertyNames(schema)
3915
4041
  // unevaluated* is generated only where it is provably its plain counterpart
3916
4042
  // (see unevalLocalOk); any occurrence that is not keeps the whole schema on
3917
4043
  // the interpreted engine's error path, exactly as before.
@@ -3974,7 +4100,7 @@ function compileToJSCodegenWithErrors(schema, schemaMap, userFormats, sourceOpts
3974
4100
  if (hasUnresolvableRef(schema, eRootDefs, eAnchors, schemaMap, new Set())) return null
3975
4101
  if (needsBaseTracking(schema, schemaMap, new Set())) return null
3976
4102
 
3977
- const ctx = { varCounter: 0, helperCode: [], rootDefs: eRootDefs, shared: [], refStack: new Set(), schemaMap: schemaMap || null, anchors: eAnchors, rootSchema: schema, userFormats: userFormats || null,
4103
+ const ctx = { varCounter: 0, helperCode: [], rootDefs: eRootDefs, shared: [], refStack: new Set(), schemaMap: schemaMap || null, anchors: eAnchors, rootSchema: inputSchema, userFormats: userFormats || null,
3978
4104
  // Custom format checkers referenced by the body, bound as
3979
4105
  // closure parameters below the way the other entry points do.
3980
4106
  closureVars: [], closureVals: [],
@@ -4159,16 +4285,26 @@ function genCodeENode(schema, v, pathExpr, lines, ctx, schemaPrefix) {
4159
4285
  if (!fnName) {
4160
4286
  fnName = '_defE' + ctx.defFns.size + '_' + defName.replace(/[^A-Za-z0-9_]/g, '_')
4161
4287
  ctx.defFns.set(defName, fnName)
4288
+ // The schema path comes in as an argument too, `_sp`, so an error
4289
+ // names the path it was reached by, as it does when the definition
4290
+ // is inlined and as the interpreted engine reports it; a fixed
4291
+ // `#/$defs/<name>` named the definition instead. The body is
4292
+ // generated against a placeholder prefix that becomes `_sp` in every
4293
+ // string it starts. One left anywhere else declines, rather than
4294
+ // emitting a path the placeholder stands in for.
4162
4295
  const bodyLines = []
4163
- genCodeE(ctx.rootDefs[defName], 'd', '_p', bodyLines, ctx, '#/$defs/' + defName)
4296
+ genCodeE(ctx.rootDefs[defName], 'd', '_p', bodyLines, ctx, SP_PLACEHOLDER)
4297
+ let body = bodyLines.join('\n ')
4298
+ body = body.split(`'${SP_PLACEHOLDER}`).join(`_sp+'`).split(`"${SP_PLACEHOLDER}`).join(`_sp+"`)
4299
+ if (body.includes(SP_PLACEHOLDER)) throw DECLINE
4164
4300
  ctx.helperCode.push(
4165
- `const ${fnName}_s=new Set()\n function ${fnName}(d,_p,_all,_e){\n ` +
4166
- `if(_sg){if(typeof d!=='object'||d===null)return ${fnName}_b(d,_p,_all,_e);if(${fnName}_s.has(d))return;${fnName}_s.add(d);try{return ${fnName}_b(d,_p,_all,_e)}finally{${fnName}_s.delete(d)}}\n ` +
4167
- `if(++_sd>${CYCLE_DEPTH})throw _CYC\n const _r=${fnName}_b(d,_p,_all,_e)\n _sd--\n return _r\n }\n ` +
4168
- `function ${fnName}_b(d,_p,_all,_e){${bodyLines.join('\n ')}}`,
4301
+ `const ${fnName}_s=new Set()\n function ${fnName}(d,_p,_all,_e,_sp){\n ` +
4302
+ `if(_sg){if(typeof d!=='object'||d===null)return ${fnName}_b(d,_p,_all,_e,_sp);if(${fnName}_s.has(d))return;${fnName}_s.add(d);try{return ${fnName}_b(d,_p,_all,_e,_sp)}finally{${fnName}_s.delete(d)}}\n ` +
4303
+ `if(++_sd>${CYCLE_DEPTH})throw _CYC\n const _r=${fnName}_b(d,_p,_all,_e,_sp)\n _sd--\n return _r\n }\n ` +
4304
+ `function ${fnName}_b(d,_p,_all,_e,_sp){${body}}`,
4169
4305
  )
4170
4306
  }
4171
- lines.push(`${fnName}(${v},${pathExpr || '""'},_all,_e);if(!_all&&_e.length)return{valid:false,errors:_e}`)
4307
+ lines.push(`${fnName}(${v},${pathExpr || '""'},_all,_e,'${schemaPrefix}');if(!_all&&_e.length)return{valid:false,errors:_e}`)
4172
4308
  return
4173
4309
  }
4174
4310
  if (ctx.refStack.has(schema.$ref)) return
@@ -4760,6 +4896,8 @@ function genIfE(schema, v, pathExpr, lines, ctx, schemaPrefix) {
4760
4896
  // Avoids double-pass (jsFn → false → errFn runs same checks again).
4761
4897
  // Uses type-aware optimizations: after type check passes, skip guards.
4762
4898
  function compileToJSCombined(schema, VALID_RESULT, schemaMap, userFormats) {
4899
+ const inputSchema = schema
4900
+ schema = normalizePropertyNames(schema)
4763
4901
  // Same provably-local rule as the error generator above.
4764
4902
  if (typeof schema === 'object' && schema !== null) {
4765
4903
  const s = JSON.stringify(schema)
@@ -4838,7 +4976,7 @@ function compileToJSCombined(schema, VALID_RESULT, schemaMap, userFormats) {
4838
4976
  if (needsBaseTracking(schema, schemaMap, new Set())) return null
4839
4977
 
4840
4978
  const ctx = { varCounter: 0, helperCode: [], shared: [], closureVars: ['_cpLen'], closureVals: [_cpLen],
4841
- rootDefs: cRootDefs, refStack: new Set(), schemaMap: schemaMap || null, anchors: cAnchors, rootSchema: schema, userFormats: userFormats || null }
4979
+ rootDefs: cRootDefs, refStack: new Set(), schemaMap: schemaMap || null, anchors: cAnchors, rootSchema: inputSchema, userFormats: userFormats || null }
4842
4980
  const lines = []
4843
4981
  try {
4844
4982
  genCodeC(schema, 'd', '', lines, ctx, '#')
@@ -5,4 +5,4 @@
5
5
  // into standalone output without a runtime `fs` read. Kept in sync by
6
6
  // `tests/test_safe_regex_source_sync.js`.
7
7
 
8
- module.exports = "'use strict'\n\n// Linear-time regex engine for JSON Schema `pattern`, used in place of JS RegExp\n// so an adversarial input cannot trigger catastrophic backtracking (ReDoS).\n//\n// It is a Pike VM: the pattern compiles to a small instruction program, and the\n// VM simulates all NFA threads in lockstep over the input, deduping by program\n// counter. Runtime is O(input * program), with no backtracking.\n//\n// Supported (the RE2 subset, which is what ata's native path also accepts):\n// literals, ., character classes, \\d \\w \\s \\D \\W \\S, anchors ^ $, quantifiers\n// * + ? {n} {n,} {n,m} (greedy or lazy, same language for a boolean test),\n// groups ( ) (?: ), alternation |. Backreferences and lookaround are not\n// supported by linear engines; compileSafe throws on them so the caller can\n// decide (ata's codegen rejects such schemas rather than risk a hang).\n\nconst WS = [[9, 13], [32, 32], [160, 160]]\nconst DIGIT = [[48, 57]]\nconst WORD = [[48, 57], [65, 90], [97, 122], [95, 95]]\n\nfunction parse (src) {\n let i = 0\n const len = src.length\n const peek = () => src[i]\n const eof = () => i >= len\n\n function parseAlt () {\n const opts = [parseConcat()]\n while (!eof() && peek() === '|') { i++; opts.push(parseConcat()) }\n return opts.length === 1 ? opts[0] : { t: 'alt', opts }\n }\n\n function parseConcat () {\n const parts = []\n while (!eof() && peek() !== '|' && peek() !== ')') parts.push(parseRepeat())\n if (parts.length === 0) return { t: 'empty' }\n return parts.length === 1 ? parts[0] : { t: 'concat', parts }\n }\n\n function parseRepeat () {\n let node = parseAtom()\n while (!eof()) {\n const ch = peek()\n if (ch === '*') { i++; node = { t: 'star', child: node } }\n else if (ch === '+') { i++; node = { t: 'plus', child: node } }\n else if (ch === '?') { i++; node = { t: 'quest', child: node } }\n else if (ch === '{') {\n const saved = i\n const q = tryQuantifier()\n if (!q) { i = saved; break }\n node = { t: 'repeat', child: node, min: q.min, max: q.max }\n } else break\n // a trailing ? makes the quantifier lazy; same language for a boolean test\n if (!eof() && peek() === '?') i++\n }\n return node\n }\n\n function tryQuantifier () {\n // assumes current char is '{'\n i++\n let min = ''\n while (!eof() && /[0-9]/.test(peek())) { min += peek(); i++ }\n if (min === '') return null\n let max\n if (peek() === '}') { i++; return { min: +min, max: +min } }\n if (peek() === ',') {\n i++\n let m = ''\n while (!eof() && /[0-9]/.test(peek())) { m += peek(); i++ }\n if (peek() !== '}') return null\n i++\n max = m === '' ? Infinity : +m\n return { min: +min, max }\n }\n return null\n }\n\n function parseAtom () {\n const ch = peek()\n if (ch === '(') {\n i++\n if (src[i] === '?') {\n if (src[i + 1] === ':') { i += 2 }\n else throw new Error('unsupported group (lookaround/named) in pattern')\n }\n const child = parseAlt()\n if (peek() !== ')') throw new Error('unbalanced ( in pattern')\n i++\n return { t: 'group', child }\n }\n if (ch === '[') return parseClass()\n if (ch === '.') { i++; return { t: 'any' } }\n if (ch === '^') { i++; return { t: 'bol' } }\n if (ch === '$') { i++; return { t: 'eol' } }\n if (ch === '\\\\') return parseEscape(false)\n if (ch === ')' || ch === '|') return { t: 'empty' }\n i++\n return { t: 'char', c: ch.charCodeAt(0) }\n }\n\n function parseClass () {\n i++ // [\n let neg = false\n if (peek() === '^') { neg = true; i++ }\n const ranges = []\n while (!eof() && peek() !== ']') {\n let lo\n if (peek() === '\\\\') {\n const esc = parseEscape(true)\n if (esc.t === 'classpart') { for (const r of esc.ranges) ranges.push(r); continue }\n lo = esc.c\n } else { lo = peek().charCodeAt(0); i++ }\n if (peek() === '-' && src[i + 1] !== ']' && i + 1 < len) {\n i++ // -\n let hi\n if (peek() === '\\\\') { const e = parseEscape(true); hi = e.c } else { hi = peek().charCodeAt(0); i++ }\n ranges.push([lo, hi])\n } else {\n ranges.push([lo, lo])\n }\n }\n if (peek() !== ']') throw new Error('unbalanced [ in pattern')\n i++\n return { t: 'class', neg, ranges }\n }\n\n function parseEscape (inClass) {\n i++ // backslash\n if (eof()) throw new Error('trailing backslash in pattern')\n const ch = peek(); i++\n switch (ch) {\n case 'd': return inClass ? { t: 'classpart', ranges: DIGIT } : { t: 'class', neg: false, ranges: DIGIT }\n case 'w': return inClass ? { t: 'classpart', ranges: WORD } : { t: 'class', neg: false, ranges: WORD }\n case 's': return inClass ? { t: 'classpart', ranges: WS } : { t: 'class', neg: false, ranges: WS }\n case 'D': if (inClass) throw new Error('\\\\D inside a class is not supported'); return { t: 'class', neg: true, ranges: DIGIT }\n case 'W': if (inClass) throw new Error('\\\\W inside a class is not supported'); return { t: 'class', neg: true, ranges: WORD }\n case 'S': if (inClass) throw new Error('\\\\S inside a class is not supported'); return { t: 'class', neg: true, ranges: WS }\n case 'n': return { t: 'char', c: 10 }\n case 'r': return { t: 'char', c: 13 }\n case 't': return { t: 'char', c: 9 }\n case 'f': return { t: 'char', c: 12 }\n case 'v': return { t: 'char', c: 11 }\n case '0': return { t: 'char', c: 0 }\n case 'x': { const h = src.slice(i, i + 2); i += 2; return { t: 'char', c: parseInt(h, 16) } }\n case 'u': { const h = src.slice(i, i + 4); i += 4; return { t: 'char', c: parseInt(h, 16) } }\n case 'b': if (inClass) return { t: 'char', c: 8 }; throw new Error('\\\\b word boundary is not supported')\n default:\n if (/[1-9]/.test(ch)) throw new Error('backreferences are not supported in pattern')\n return { t: 'char', c: ch.charCodeAt(0) }\n }\n }\n\n const ast = parseAlt()\n if (!eof()) throw new Error('unexpected \"' + peek() + '\" in pattern')\n return ast\n}\n\nfunction compileProg (ast) {\n const prog = []\n const emit = (op, extra) => { const idx = prog.length; prog.push(Object.assign({ op }, extra)); return idx }\n\n function rec (n) {\n switch (n.t) {\n case 'empty': break\n case 'char': emit('char', { c: n.c }); break\n case 'any': emit('any'); break\n case 'class': emit('class', { neg: n.neg, ranges: n.ranges }); break\n case 'bol': emit('bol'); break\n case 'eol': emit('eol'); break\n case 'group': rec(n.child); break\n case 'concat': for (const p of n.parts) rec(p); break\n case 'alt': {\n const jmps = []\n for (let k = 0; k < n.opts.length; k++) {\n if (k < n.opts.length - 1) {\n const sp = emit('split', { x: 0, y: 0 })\n prog[sp].x = prog.length\n rec(n.opts[k])\n jmps.push(emit('jmp', { x: 0 }))\n prog[sp].y = prog.length\n } else {\n rec(n.opts[k])\n }\n }\n for (const j of jmps) prog[j].x = prog.length\n break\n }\n case 'star': {\n const sp = emit('split', { x: 0, y: 0 })\n prog[sp].x = prog.length\n rec(n.child)\n emit('jmp', { x: sp })\n prog[sp].y = prog.length\n break\n }\n case 'plus': {\n const start = prog.length\n rec(n.child)\n const sp = emit('split', { x: start, y: 0 })\n prog[sp].y = prog.length\n break\n }\n case 'quest': {\n const sp = emit('split', { x: 0, y: 0 })\n prog[sp].x = prog.length\n rec(n.child)\n prog[sp].y = prog.length\n break\n }\n case 'repeat': {\n for (let k = 0; k < n.min; k++) rec(n.child)\n if (n.max === Infinity) {\n if (n.min === 0) rec({ t: 'star', child: n.child })\n else rec({ t: 'star', child: n.child })\n } else {\n for (let k = 0; k < n.max - n.min; k++) rec({ t: 'quest', child: n.child })\n }\n break\n }\n }\n }\n\n rec(ast)\n emit('match')\n return prog\n}\n\n// Numeric opcodes for the runner. The program is compiled once into flat\n// typed arrays so the inner loop does no property lookups or string compares.\nconst OP_CHAR = 0\nconst OP_ANY = 1\nconst OP_CLASS = 2\nconst OP_SPLIT = 3\nconst OP_JMP = 4\nconst OP_BOL = 5\nconst OP_EOL = 6\nconst OP_MATCH = 7\n\nfunction classMatcher (instr) {\n // ASCII is answered from a bitmap; anything above 0x7f walks the ranges.\n const bits = new Uint8Array(128)\n const r = instr.ranges\n for (let k = 0; k < r.length; k++) {\n const hi = Math.min(r[k][1], 127)\n for (let c = r[k][0]; c <= hi; c++) bits[c] = 1\n }\n return { bits, ranges: r, neg: instr.neg }\n}\n\nfunction matchClass (cls, c) {\n let inside\n if (c < 128) {\n inside = cls.bits[c] === 1\n } else {\n inside = false\n const r = cls.ranges\n for (let k = 0; k < r.length; k++) { if (c >= r[k][0] && c <= r[k][1]) { inside = true; break } }\n }\n return cls.neg ? !inside : inside\n}\n\nfunction makeRunner (prog) {\n const n = prog.length\n const ops = new Uint8Array(n)\n const xs = new Int32Array(n)\n const ys = new Int32Array(n)\n const cs = new Int32Array(n)\n const classes = new Array(n)\n for (let i = 0; i < n; i++) {\n const I = prog[i]\n switch (I.op) {\n case 'char': ops[i] = OP_CHAR; cs[i] = I.c; break\n case 'any': ops[i] = OP_ANY; break\n case 'class': ops[i] = OP_CLASS; classes[i] = classMatcher(I); break\n case 'split': ops[i] = OP_SPLIT; xs[i] = I.x; ys[i] = I.y; break\n case 'jmp': ops[i] = OP_JMP; xs[i] = I.x; break\n case 'bol': ops[i] = OP_BOL; break\n case 'eol': ops[i] = OP_EOL; break\n case 'match': ops[i] = OP_MATCH; break\n }\n }\n\n const lastGen = new Int32Array(n).fill(-1)\n let gen = 0\n // Each unvisited instruction is popped once and pushes at most two, so the\n // stack never holds more than 2n + 1 entries.\n const stack = new Int32Array(2 * n + 2)\n // Thread lists hold at most one entry per instruction per step.\n let clist = new Int32Array(n)\n let nlist = new Int32Array(n)\n let clen = 0\n let nlen = 0\n\n // Follows epsilon edges from `pc` and records every consuming instruction\n // (or match) reached in `list`. `lastGen` dedupes per step.\n function addThread (list, len0, pc, pos, len) {\n // Most transitions land directly on a consuming instruction; skip the\n // stack walk for those.\n if (ops[pc] <= OP_CLASS || ops[pc] === OP_MATCH) {\n if (lastGen[pc] === gen) return len0\n lastGen[pc] = gen\n list[len0] = pc\n return len0 + 1\n }\n let sp = 0\n stack[sp++] = pc\n let count = len0\n while (sp > 0) {\n const p = stack[--sp]\n if (lastGen[p] === gen) continue\n lastGen[p] = gen\n switch (ops[p]) {\n case OP_JMP: stack[sp++] = xs[p]; break\n case OP_SPLIT: stack[sp++] = ys[p]; stack[sp++] = xs[p]; break\n case OP_BOL: if (pos === 0) stack[sp++] = p + 1; break\n case OP_EOL: if (pos === len) stack[sp++] = p + 1; break\n default: list[count++] = p\n }\n }\n return count\n }\n\n // A pattern is anchored when starting it anywhere but position 0 yields no\n // thread, which is the case for `^...` and its alternations. The probe sits\n // at position 1 of a length-1 string so that only `^` can fail. For anchored\n // patterns the per-position restart below is skipped, and an empty thread\n // list means the match has already failed.\n gen++\n const anchored = addThread(nlist, 0, 0, 1, 1) === 0\n\n function testNFA (s) {\n const len = s.length\n gen++\n clen = addThread(clist, 0, 0, 0, len)\n for (let pos = 0; pos <= len; pos++) {\n const c = pos < len ? s.charCodeAt(pos) : -1\n gen++\n nlen = 0\n for (let k = 0; k < clen; k++) {\n const pc = clist[k]\n switch (ops[pc]) {\n case OP_MATCH: return true\n case OP_CHAR: if (c === cs[pc]) nlen = addThread(nlist, nlen, pc + 1, pos + 1, len); break\n case OP_ANY: if (c !== -1 && c !== 10) nlen = addThread(nlist, nlen, pc + 1, pos + 1, len); break\n case OP_CLASS: if (c !== -1 && matchClass(classes[pc], c)) nlen = addThread(nlist, nlen, pc + 1, pos + 1, len); break\n }\n }\n if (pos < len) {\n if (!anchored) nlen = addThread(nlist, nlen, 0, pos + 1, len)\n else if (nlen === 0) return false\n }\n const tmp = clist; clist = nlist; nlist = tmp\n clen = nlen\n }\n return false\n }\n\n // Lazy DFA on top of the NFA. A DFA state is the set of consuming\n // instructions live at a position; transitions are computed on first use\n // and cached per ASCII character. `^` and `$` depend on position, so the\n // closure is taken with flags for \"at start\" and \"at end\", which gives two\n // start states and two transition tables per state. The state count is\n // capped; past the cap the matcher falls back to the NFA walk above, so the\n // time bound stays linear either way.\n const MAX_STATES = 256\n const states = []\n const stateIds = new Map()\n let overflow = false\n\n function closure (list, count, pc, atStart, atEnd) {\n // Same walk as addThread, with the position replaced by the two flags.\n let sp = 0\n stack[sp++] = pc\n while (sp > 0) {\n const p = stack[--sp]\n if (lastGen[p] === gen) continue\n lastGen[p] = gen\n switch (ops[p]) {\n case OP_JMP: stack[sp++] = xs[p]; break\n case OP_SPLIT: stack[sp++] = ys[p]; stack[sp++] = xs[p]; break\n case OP_BOL: if (atStart) stack[sp++] = p + 1; break\n case OP_EOL: if (atEnd) stack[sp++] = p + 1; break\n default: list[count++] = p\n }\n }\n return count\n }\n\n function internState (list, count) {\n const pcs = Array.from(list.subarray(0, count)).sort((a, b) => a - b)\n const key = pcs.join(',')\n let id = stateIds.get(key)\n if (id !== undefined) return id\n if (states.length >= MAX_STATES) { overflow = true; return -1 }\n id = states.length\n let isMatch = false\n for (let k = 0; k < pcs.length; k++) if (ops[pcs[k]] === OP_MATCH) { isMatch = true; break }\n states.push({ pcs: Int32Array.from(pcs), isMatch, next: new Int32Array(128).fill(-2), nextEnd: new Int32Array(128).fill(-2) })\n stateIds.set(key, id)\n return id\n }\n\n function step (state, c, atEnd) {\n gen++\n let count = 0\n const pcs = state.pcs\n for (let k = 0; k < pcs.length; k++) {\n const pc = pcs[k]\n switch (ops[pc]) {\n case OP_CHAR: if (c === cs[pc]) count = closure(nlist, count, pc + 1, false, atEnd); break\n case OP_ANY: if (c !== 10) count = closure(nlist, count, pc + 1, false, atEnd); break\n case OP_CLASS: if (matchClass(classes[pc], c)) count = closure(nlist, count, pc + 1, false, atEnd); break\n }\n }\n if (!anchored) count = closure(nlist, count, 0, false, atEnd)\n return internState(nlist, count)\n }\n\n let startEmpty = -2\n let startNonEmpty = -2\n\n function startState (atEnd) {\n gen++\n const count = closure(nlist, 0, 0, true, atEnd)\n return internState(nlist, count)\n }\n\n function testDFA (s) {\n const len = s.length\n let id\n if (len === 0) {\n if (startEmpty === -2) startEmpty = startState(true)\n id = startEmpty\n } else {\n if (startNonEmpty === -2) startNonEmpty = startState(false)\n id = startNonEmpty\n }\n if (id < 0) return testNFA(s)\n let state = states[id]\n for (let pos = 0; pos < len; pos++) {\n if (state.isMatch) return true\n const c = s.charCodeAt(pos)\n const atEnd = pos + 1 === len\n let nid\n if (c < 128) {\n const table = atEnd ? state.nextEnd : state.next\n nid = table[c]\n if (nid === -2) { nid = step(state, c, atEnd); table[c] = nid }\n } else {\n nid = step(state, c, atEnd)\n }\n if (nid < 0) return testNFA(s)\n state = states[nid]\n if (anchored && state.pcs.length === 0) return false\n }\n return state.isMatch\n }\n\n return function test (s) {\n return overflow ? testNFA(s) : testDFA(s)\n }\n}\n\nfunction compileSafe (pattern) {\n const prog = compileProg(parse(pattern))\n const runner = makeRunner(prog)\n // `__ataSafe` brands the result so the standalone serializer can tell a safe\n // matcher apart from a RegExp and emit `__ataSafeRe(source)` instead.\n return { test: runner, source: pattern, __ataSafe: true }\n}\n\n// True when the linear engine can represent `src`. Used by the codegen to decide\n// between the safe matcher and a JS RegExp fallback for patterns outside the\n// supported (RE2) subset (backreferences, lookaround, etc.).\nfunction patternIsSafe (src) {\n try { compileSafe(src); return true } catch { return false }\n}\n\nmodule.exports = { compileSafe, patternIsSafe }\n";
8
+ module.exports = "'use strict'\n\n// Linear-time regex engine for JSON Schema `pattern`, used in place of JS RegExp\n// so an adversarial input cannot trigger catastrophic backtracking (ReDoS).\n//\n// It is a Pike VM: the pattern compiles to a small instruction program, and the\n// VM simulates all NFA threads in lockstep over the input, deduping by program\n// counter. Runtime is O(input * program), with no backtracking.\n//\n// Supported (the RE2 subset, which is what ata's native path also accepts):\n// literals, ., character classes, \\d \\w \\s \\D \\W \\S, anchors ^ $, quantifiers\n// * + ? {n} {n,} {n,m} (greedy or lazy, same language for a boolean test),\n// groups ( ) (?: ), alternation |. Backreferences and lookaround are not\n// supported by linear engines; compileSafe throws on them so the caller can\n// decide (ata's codegen rejects such schemas rather than risk a hang).\n\n// ECMA-262 WhiteSpace and LineTerminator, what `\\s` matches: tab, line feed,\n// vertical tab, form feed, carriage return, space, no-break space, the\n// Unicode space separators, the line and paragraph separators, and the byte\n// order mark. The set used to stop at U+00A0, so `^\\S+$` accepted a string\n// holding U+2028 or U+3000.\nconst WS = [[9, 13], [32, 32], [160, 160], [0x1680, 0x1680], [0x2000, 0x200a], [0x2028, 0x2029], [0x202f, 0x202f], [0x205f, 0x205f], [0x3000, 0x3000], [0xfeff, 0xfeff]]\n// `.` matches anything but a line terminator: LF, CR and U+2028/U+2029. It\n// excluded only LF, so `^.$` accepted \"\\r\".\nconst isLineTerminator = (c) => c === 10 || c === 13 || c === 0x2028 || c === 0x2029\nconst DIGIT = [[48, 57]]\nconst WORD = [[48, 57], [65, 90], [97, 122], [95, 95]]\n\nfunction parse (src) {\n let i = 0\n const len = src.length\n const peek = () => src[i]\n const eof = () => i >= len\n\n function parseAlt () {\n const opts = [parseConcat()]\n while (!eof() && peek() === '|') { i++; opts.push(parseConcat()) }\n return opts.length === 1 ? opts[0] : { t: 'alt', opts }\n }\n\n function parseConcat () {\n const parts = []\n while (!eof() && peek() !== '|' && peek() !== ')') parts.push(parseRepeat())\n if (parts.length === 0) return { t: 'empty' }\n return parts.length === 1 ? parts[0] : { t: 'concat', parts }\n }\n\n function parseRepeat () {\n let node = parseAtom()\n while (!eof()) {\n const ch = peek()\n if (ch === '*') { i++; node = { t: 'star', child: node } }\n else if (ch === '+') { i++; node = { t: 'plus', child: node } }\n else if (ch === '?') { i++; node = { t: 'quest', child: node } }\n else if (ch === '{') {\n const saved = i\n const q = tryQuantifier()\n if (!q) { i = saved; break }\n node = { t: 'repeat', child: node, min: q.min, max: q.max }\n } else break\n // a trailing ? makes the quantifier lazy; same language for a boolean test\n if (!eof() && peek() === '?') i++\n }\n return node\n }\n\n function tryQuantifier () {\n // assumes current char is '{'\n i++\n let min = ''\n while (!eof() && /[0-9]/.test(peek())) { min += peek(); i++ }\n if (min === '') return null\n let max\n if (peek() === '}') { i++; return { min: +min, max: +min } }\n if (peek() === ',') {\n i++\n let m = ''\n while (!eof() && /[0-9]/.test(peek())) { m += peek(); i++ }\n if (peek() !== '}') return null\n i++\n max = m === '' ? Infinity : +m\n return { min: +min, max }\n }\n return null\n }\n\n function parseAtom () {\n const ch = peek()\n if (ch === '(') {\n i++\n if (src[i] === '?') {\n if (src[i + 1] === ':') { i += 2 }\n else throw new Error('unsupported group (lookaround/named) in pattern')\n }\n const child = parseAlt()\n if (peek() !== ')') throw new Error('unbalanced ( in pattern')\n i++\n return { t: 'group', child }\n }\n if (ch === '[') return parseClass()\n if (ch === '.') { i++; return { t: 'any' } }\n if (ch === '^') { i++; return { t: 'bol' } }\n if (ch === '$') { i++; return { t: 'eol' } }\n if (ch === '\\\\') return parseEscape(false)\n if (ch === ')' || ch === '|') return { t: 'empty' }\n i++\n return { t: 'char', c: ch.charCodeAt(0) }\n }\n\n function parseClass () {\n i++ // [\n let neg = false\n if (peek() === '^') { neg = true; i++ }\n const ranges = []\n while (!eof() && peek() !== ']') {\n let lo\n if (peek() === '\\\\') {\n const esc = parseEscape(true)\n if (esc.t === 'classpart') { for (const r of esc.ranges) ranges.push(r); continue }\n lo = esc.c\n } else { lo = peek().charCodeAt(0); i++ }\n if (peek() === '-' && src[i + 1] !== ']' && i + 1 < len) {\n i++ // -\n let hi\n if (peek() === '\\\\') { const e = parseEscape(true); hi = e.c } else { hi = peek().charCodeAt(0); i++ }\n ranges.push([lo, hi])\n } else {\n ranges.push([lo, lo])\n }\n }\n if (peek() !== ']') throw new Error('unbalanced [ in pattern')\n i++\n return { t: 'class', neg, ranges }\n }\n\n function parseEscape (inClass) {\n i++ // backslash\n if (eof()) throw new Error('trailing backslash in pattern')\n const ch = peek(); i++\n switch (ch) {\n case 'd': return inClass ? { t: 'classpart', ranges: DIGIT } : { t: 'class', neg: false, ranges: DIGIT }\n case 'w': return inClass ? { t: 'classpart', ranges: WORD } : { t: 'class', neg: false, ranges: WORD }\n case 's': return inClass ? { t: 'classpart', ranges: WS } : { t: 'class', neg: false, ranges: WS }\n case 'D': if (inClass) throw new Error('\\\\D inside a class is not supported'); return { t: 'class', neg: true, ranges: DIGIT }\n case 'W': if (inClass) throw new Error('\\\\W inside a class is not supported'); return { t: 'class', neg: true, ranges: WORD }\n case 'S': if (inClass) throw new Error('\\\\S inside a class is not supported'); return { t: 'class', neg: true, ranges: WS }\n case 'n': return { t: 'char', c: 10 }\n case 'r': return { t: 'char', c: 13 }\n case 't': return { t: 'char', c: 9 }\n case 'f': return { t: 'char', c: 12 }\n case 'v': return { t: 'char', c: 11 }\n case '0': return { t: 'char', c: 0 }\n case 'x': { const h = src.slice(i, i + 2); i += 2; return { t: 'char', c: parseInt(h, 16) } }\n case 'u': { const h = src.slice(i, i + 4); i += 4; return { t: 'char', c: parseInt(h, 16) } }\n case 'b': if (inClass) return { t: 'char', c: 8 }; throw new Error('\\\\b word boundary is not supported')\n default:\n if (/[1-9]/.test(ch)) throw new Error('backreferences are not supported in pattern')\n return { t: 'char', c: ch.charCodeAt(0) }\n }\n }\n\n const ast = parseAlt()\n if (!eof()) throw new Error('unexpected \"' + peek() + '\" in pattern')\n return ast\n}\n\nfunction compileProg (ast) {\n const prog = []\n const emit = (op, extra) => { const idx = prog.length; prog.push(Object.assign({ op }, extra)); return idx }\n\n function rec (n) {\n switch (n.t) {\n case 'empty': break\n case 'char': emit('char', { c: n.c }); break\n case 'any': emit('any'); break\n case 'class': emit('class', { neg: n.neg, ranges: n.ranges }); break\n case 'bol': emit('bol'); break\n case 'eol': emit('eol'); break\n case 'group': rec(n.child); break\n case 'concat': for (const p of n.parts) rec(p); break\n case 'alt': {\n const jmps = []\n for (let k = 0; k < n.opts.length; k++) {\n if (k < n.opts.length - 1) {\n const sp = emit('split', { x: 0, y: 0 })\n prog[sp].x = prog.length\n rec(n.opts[k])\n jmps.push(emit('jmp', { x: 0 }))\n prog[sp].y = prog.length\n } else {\n rec(n.opts[k])\n }\n }\n for (const j of jmps) prog[j].x = prog.length\n break\n }\n case 'star': {\n const sp = emit('split', { x: 0, y: 0 })\n prog[sp].x = prog.length\n rec(n.child)\n emit('jmp', { x: sp })\n prog[sp].y = prog.length\n break\n }\n case 'plus': {\n const start = prog.length\n rec(n.child)\n const sp = emit('split', { x: start, y: 0 })\n prog[sp].y = prog.length\n break\n }\n case 'quest': {\n const sp = emit('split', { x: 0, y: 0 })\n prog[sp].x = prog.length\n rec(n.child)\n prog[sp].y = prog.length\n break\n }\n case 'repeat': {\n for (let k = 0; k < n.min; k++) rec(n.child)\n if (n.max === Infinity) {\n if (n.min === 0) rec({ t: 'star', child: n.child })\n else rec({ t: 'star', child: n.child })\n } else {\n for (let k = 0; k < n.max - n.min; k++) rec({ t: 'quest', child: n.child })\n }\n break\n }\n }\n }\n\n rec(ast)\n emit('match')\n return prog\n}\n\n// Numeric opcodes for the runner. The program is compiled once into flat\n// typed arrays so the inner loop does no property lookups or string compares.\nconst OP_CHAR = 0\nconst OP_ANY = 1\nconst OP_CLASS = 2\nconst OP_SPLIT = 3\nconst OP_JMP = 4\nconst OP_BOL = 5\nconst OP_EOL = 6\nconst OP_MATCH = 7\n\nfunction classMatcher (instr) {\n // ASCII is answered from a bitmap; anything above 0x7f walks the ranges.\n const bits = new Uint8Array(128)\n const r = instr.ranges\n for (let k = 0; k < r.length; k++) {\n const hi = Math.min(r[k][1], 127)\n for (let c = r[k][0]; c <= hi; c++) bits[c] = 1\n }\n return { bits, ranges: r, neg: instr.neg }\n}\n\nfunction matchClass (cls, c) {\n let inside\n if (c < 128) {\n inside = cls.bits[c] === 1\n } else {\n inside = false\n const r = cls.ranges\n for (let k = 0; k < r.length; k++) { if (c >= r[k][0] && c <= r[k][1]) { inside = true; break } }\n }\n return cls.neg ? !inside : inside\n}\n\nfunction makeRunner (prog) {\n const n = prog.length\n const ops = new Uint8Array(n)\n const xs = new Int32Array(n)\n const ys = new Int32Array(n)\n const cs = new Int32Array(n)\n const classes = new Array(n)\n for (let i = 0; i < n; i++) {\n const I = prog[i]\n switch (I.op) {\n case 'char': ops[i] = OP_CHAR; cs[i] = I.c; break\n case 'any': ops[i] = OP_ANY; break\n case 'class': ops[i] = OP_CLASS; classes[i] = classMatcher(I); break\n case 'split': ops[i] = OP_SPLIT; xs[i] = I.x; ys[i] = I.y; break\n case 'jmp': ops[i] = OP_JMP; xs[i] = I.x; break\n case 'bol': ops[i] = OP_BOL; break\n case 'eol': ops[i] = OP_EOL; break\n case 'match': ops[i] = OP_MATCH; break\n }\n }\n\n const lastGen = new Int32Array(n).fill(-1)\n let gen = 0\n // Each unvisited instruction is popped once and pushes at most two, so the\n // stack never holds more than 2n + 1 entries.\n const stack = new Int32Array(2 * n + 2)\n // Thread lists hold at most one entry per instruction per step.\n let clist = new Int32Array(n)\n let nlist = new Int32Array(n)\n let clen = 0\n let nlen = 0\n\n // Follows epsilon edges from `pc` and records every consuming instruction\n // (or match) reached in `list`. `lastGen` dedupes per step.\n function addThread (list, len0, pc, pos, len) {\n // Most transitions land directly on a consuming instruction; skip the\n // stack walk for those.\n if (ops[pc] <= OP_CLASS || ops[pc] === OP_MATCH) {\n if (lastGen[pc] === gen) return len0\n lastGen[pc] = gen\n list[len0] = pc\n return len0 + 1\n }\n let sp = 0\n stack[sp++] = pc\n let count = len0\n while (sp > 0) {\n const p = stack[--sp]\n if (lastGen[p] === gen) continue\n lastGen[p] = gen\n switch (ops[p]) {\n case OP_JMP: stack[sp++] = xs[p]; break\n case OP_SPLIT: stack[sp++] = ys[p]; stack[sp++] = xs[p]; break\n case OP_BOL: if (pos === 0) stack[sp++] = p + 1; break\n case OP_EOL: if (pos === len) stack[sp++] = p + 1; break\n default: list[count++] = p\n }\n }\n return count\n }\n\n // A pattern is anchored when starting it anywhere but position 0 yields no\n // thread, which is the case for `^...` and its alternations. The probe sits\n // at position 1 of a length-1 string so that only `^` can fail. For anchored\n // patterns the per-position restart below is skipped, and an empty thread\n // list means the match has already failed.\n gen++\n const anchored = addThread(nlist, 0, 0, 1, 1) === 0\n\n function testNFA (s) {\n const len = s.length\n gen++\n clen = addThread(clist, 0, 0, 0, len)\n for (let pos = 0; pos <= len; pos++) {\n const c = pos < len ? s.charCodeAt(pos) : -1\n gen++\n nlen = 0\n for (let k = 0; k < clen; k++) {\n const pc = clist[k]\n switch (ops[pc]) {\n case OP_MATCH: return true\n case OP_CHAR: if (c === cs[pc]) nlen = addThread(nlist, nlen, pc + 1, pos + 1, len); break\n case OP_ANY: if (c !== -1 && !isLineTerminator(c)) nlen = addThread(nlist, nlen, pc + 1, pos + 1, len); break\n case OP_CLASS: if (c !== -1 && matchClass(classes[pc], c)) nlen = addThread(nlist, nlen, pc + 1, pos + 1, len); break\n }\n }\n if (pos < len) {\n if (!anchored) nlen = addThread(nlist, nlen, 0, pos + 1, len)\n else if (nlen === 0) return false\n }\n const tmp = clist; clist = nlist; nlist = tmp\n clen = nlen\n }\n return false\n }\n\n // Lazy DFA on top of the NFA. A DFA state is the set of consuming\n // instructions live at a position; transitions are computed on first use\n // and cached per ASCII character. `^` and `$` depend on position, so the\n // closure is taken with flags for \"at start\" and \"at end\", which gives two\n // start states and two transition tables per state. The state count is\n // capped; past the cap the matcher falls back to the NFA walk above, so the\n // time bound stays linear either way.\n const MAX_STATES = 256\n const states = []\n const stateIds = new Map()\n let overflow = false\n\n function closure (list, count, pc, atStart, atEnd) {\n // Same walk as addThread, with the position replaced by the two flags.\n let sp = 0\n stack[sp++] = pc\n while (sp > 0) {\n const p = stack[--sp]\n if (lastGen[p] === gen) continue\n lastGen[p] = gen\n switch (ops[p]) {\n case OP_JMP: stack[sp++] = xs[p]; break\n case OP_SPLIT: stack[sp++] = ys[p]; stack[sp++] = xs[p]; break\n case OP_BOL: if (atStart) stack[sp++] = p + 1; break\n case OP_EOL: if (atEnd) stack[sp++] = p + 1; break\n default: list[count++] = p\n }\n }\n return count\n }\n\n function internState (list, count) {\n const pcs = Array.from(list.subarray(0, count)).sort((a, b) => a - b)\n const key = pcs.join(',')\n let id = stateIds.get(key)\n if (id !== undefined) return id\n if (states.length >= MAX_STATES) { overflow = true; return -1 }\n id = states.length\n let isMatch = false\n for (let k = 0; k < pcs.length; k++) if (ops[pcs[k]] === OP_MATCH) { isMatch = true; break }\n states.push({ pcs: Int32Array.from(pcs), isMatch, next: new Int32Array(128).fill(-2), nextEnd: new Int32Array(128).fill(-2) })\n stateIds.set(key, id)\n return id\n }\n\n function step (state, c, atEnd) {\n gen++\n let count = 0\n const pcs = state.pcs\n for (let k = 0; k < pcs.length; k++) {\n const pc = pcs[k]\n switch (ops[pc]) {\n case OP_CHAR: if (c === cs[pc]) count = closure(nlist, count, pc + 1, false, atEnd); break\n case OP_ANY: if (!isLineTerminator(c)) count = closure(nlist, count, pc + 1, false, atEnd); break\n case OP_CLASS: if (matchClass(classes[pc], c)) count = closure(nlist, count, pc + 1, false, atEnd); break\n }\n }\n if (!anchored) count = closure(nlist, count, 0, false, atEnd)\n return internState(nlist, count)\n }\n\n let startEmpty = -2\n let startNonEmpty = -2\n\n function startState (atEnd) {\n gen++\n const count = closure(nlist, 0, 0, true, atEnd)\n return internState(nlist, count)\n }\n\n function testDFA (s) {\n const len = s.length\n let id\n if (len === 0) {\n if (startEmpty === -2) startEmpty = startState(true)\n id = startEmpty\n } else {\n if (startNonEmpty === -2) startNonEmpty = startState(false)\n id = startNonEmpty\n }\n if (id < 0) return testNFA(s)\n let state = states[id]\n for (let pos = 0; pos < len; pos++) {\n if (state.isMatch) return true\n const c = s.charCodeAt(pos)\n const atEnd = pos + 1 === len\n let nid\n if (c < 128) {\n const table = atEnd ? state.nextEnd : state.next\n nid = table[c]\n if (nid === -2) { nid = step(state, c, atEnd); table[c] = nid }\n } else {\n nid = step(state, c, atEnd)\n }\n if (nid < 0) return testNFA(s)\n state = states[nid]\n if (anchored && state.pcs.length === 0) return false\n }\n return state.isMatch\n }\n\n return function test (s) {\n return overflow ? testNFA(s) : testDFA(s)\n }\n}\n\nfunction compileSafe (pattern) {\n const prog = compileProg(parse(pattern))\n const runner = makeRunner(prog)\n // `__ataSafe` brands the result so the standalone serializer can tell a safe\n // matcher apart from a RegExp and emit `__ataSafeRe(source)` instead.\n return { test: runner, source: pattern, __ataSafe: true }\n}\n\n// True when the linear engine can represent `src`. Used by the codegen to decide\n// between the safe matcher and a JS RegExp fallback for patterns outside the\n// supported (RE2) subset (backreferences, lookaround, etc.).\nfunction patternIsSafe (src) {\n try { compileSafe(src); return true } catch { return false }\n}\n\nmodule.exports = { compileSafe, patternIsSafe }\n";
package/lib/safe-regex.js CHANGED
@@ -14,7 +14,15 @@
14
14
  // supported by linear engines; compileSafe throws on them so the caller can
15
15
  // decide (ata's codegen rejects such schemas rather than risk a hang).
16
16
 
17
- const WS = [[9, 13], [32, 32], [160, 160]]
17
+ // ECMA-262 WhiteSpace and LineTerminator, what `\s` matches: tab, line feed,
18
+ // vertical tab, form feed, carriage return, space, no-break space, the
19
+ // Unicode space separators, the line and paragraph separators, and the byte
20
+ // order mark. The set used to stop at U+00A0, so `^\S+$` accepted a string
21
+ // holding U+2028 or U+3000.
22
+ const WS = [[9, 13], [32, 32], [160, 160], [0x1680, 0x1680], [0x2000, 0x200a], [0x2028, 0x2029], [0x202f, 0x202f], [0x205f, 0x205f], [0x3000, 0x3000], [0xfeff, 0xfeff]]
23
+ // `.` matches anything but a line terminator: LF, CR and U+2028/U+2029. It
24
+ // excluded only LF, so `^.$` accepted "\r".
25
+ const isLineTerminator = (c) => c === 10 || c === 13 || c === 0x2028 || c === 0x2029
18
26
  const DIGIT = [[48, 57]]
19
27
  const WORD = [[48, 57], [65, 90], [97, 122], [95, 95]]
20
28
 
@@ -342,7 +350,7 @@ function makeRunner (prog) {
342
350
  switch (ops[pc]) {
343
351
  case OP_MATCH: return true
344
352
  case OP_CHAR: if (c === cs[pc]) nlen = addThread(nlist, nlen, pc + 1, pos + 1, len); break
345
- case OP_ANY: if (c !== -1 && c !== 10) nlen = addThread(nlist, nlen, pc + 1, pos + 1, len); break
353
+ case OP_ANY: if (c !== -1 && !isLineTerminator(c)) nlen = addThread(nlist, nlen, pc + 1, pos + 1, len); break
346
354
  case OP_CLASS: if (c !== -1 && matchClass(classes[pc], c)) nlen = addThread(nlist, nlen, pc + 1, pos + 1, len); break
347
355
  }
348
356
  }
@@ -409,7 +417,7 @@ function makeRunner (prog) {
409
417
  const pc = pcs[k]
410
418
  switch (ops[pc]) {
411
419
  case OP_CHAR: if (c === cs[pc]) count = closure(nlist, count, pc + 1, false, atEnd); break
412
- case OP_ANY: if (c !== 10) count = closure(nlist, count, pc + 1, false, atEnd); break
420
+ case OP_ANY: if (!isLineTerminator(c)) count = closure(nlist, count, pc + 1, false, atEnd); break
413
421
  case OP_CLASS: if (matchClass(classes[pc], c)) count = closure(nlist, count, pc + 1, false, atEnd); break
414
422
  }
415
423
  }
@@ -1471,6 +1471,7 @@ class Validator {
1471
1471
  value: Object.freeze({
1472
1472
  version: 1,
1473
1473
  vendor: "ata-validator",
1474
+ jsonSchema: standardJsonSchema(() => schemaObj),
1474
1475
  validate(value) {
1475
1476
  const result = v.validate(value);
1476
1477
  if (result.valid) return { value };
@@ -1612,6 +1613,31 @@ function defineSchema (schema) {
1612
1613
  // bound closure, stores it on the instance as an ordinary writable property
1613
1614
  // and returns it. The setter keeps the compile step's plain assignments
1614
1615
  // (`this.validate = fn`) working before the getter has ever run. Detached
1616
+ // Standard JSON Schema, the `jsonSchema` member of `~standard`: what a consumer
1617
+ // such as the MCP SDK reads to publish a schema, from `input({ target })`.
1618
+ // ata validates the JSON Schema it was given and does not transform values, so
1619
+ // input and output are the same document, handed back as a copy when the
1620
+ // target is the schema's dialect: draft-07 when it declares draft-07, 2020-12
1621
+ // when it declares 2020-12 or nothing, as ata reads it. Any other target
1622
+ // throws, as the specification asks, rather than returning a document that
1623
+ // would mean something else there.
1624
+ function standardJsonSchema(getSchema) {
1625
+ const convert = (options) => {
1626
+ const target = options && options.target;
1627
+ const schema = getSchema();
1628
+ if (schema === true) return {};
1629
+ if (schema === false) return { not: {} };
1630
+ const declared = typeof schema.$schema === 'string' ? schema.$schema : '';
1631
+ const dialect = declared === '' || declared.includes('draft/2020-12/schema') ? 'draft-2020-12'
1632
+ : declared.includes('draft-07/schema') ? 'draft-07' : declared;
1633
+ if (target !== dialect) {
1634
+ throw new TypeError(`ata-validator cannot provide a ${dialect} schema as ${String(target)}; it does not convert between dialects`);
1635
+ }
1636
+ return JSON.parse(JSON.stringify(schema));
1637
+ };
1638
+ return Object.freeze({ input: convert, output: convert });
1639
+ }
1640
+
1615
1641
  // use (`const f = v.validate`) keeps working because the closure binds the
1616
1642
  // instance.
1617
1643
  // Standard Schema V1. Built on first read, then pinned to the instance with
@@ -1623,6 +1649,7 @@ Object.defineProperty(Validator.prototype, "~standard", {
1623
1649
  const std = Object.freeze({
1624
1650
  version: 1,
1625
1651
  vendor: "ata-validator",
1652
+ jsonSchema: standardJsonSchema(() => self._rawSchema),
1626
1653
  validate(value) {
1627
1654
  const result = self.validate(value);
1628
1655
  if (result.valid) {
package/lib/version.js CHANGED
@@ -7,4 +7,4 @@
7
7
  //
8
8
  // Kept in lockstep with package.json by `tests/test_version_sync.js`.
9
9
 
10
- module.exports = '1.36.1';
10
+ module.exports = '1.37.1';
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "ata-validator",
3
- "version": "1.36.1",
3
+ "version": "1.37.1",
4
4
  "description": "JSON Schema validation that compiles for speed and still runs where code generation is blocked. Compiled and interpreted engines answer identically at 100% of the official suite. TypeScript inference, Standard Schema V1, and a build step that emits dependency-free modules.",
5
5
  "main": "index.js",
6
6
  "module": "index.mjs",
@@ -59,7 +59,7 @@
59
59
  "release:check": "node scripts/regen-safe-regex-source.js && node tests/test_pack_purity.js && node scripts/check-doc-coverage.js && node tests/test_error_codes_lock.js && node tests/test_safe_regex_source_sync.js && node tests/test_version_sync.js",
60
60
  "build": "cmake-js build --target ata",
61
61
  "rebuild": "cmake-js rebuild --target ata",
62
- "test": "node test.js && node tests/test_removed_aot_methods.js && node tests/test_no_native.js && node tests/test_no_eval.js && node tests/test_property_dependencies.js && node tests/test_v1_dialect.js && node tests/test_buffer_path_parity.js && node tests/test_text_path_errors.js && node tests/test_compiled_parity.js && node tests/test_buffer_gate.js && node tests/test_scanner_differential.js && node tests/test_validate_json_scanner_cost.js && node tests/test_buffer_reject_cost.js && node tests/test_nan_verdict.js && node tests/test_remove_additional_nested.js && node tests/test_exclusive_bounds.js && node tests/test_engine_differential.js && node tests/test_error_shape_differential.js && node tests/test_output_format.js && node tests/test_retry_message.js && node tests/test_describe_schema.js && node tests/test_aot_parse.js && node tests/test_draft7_semantics.js && node tests/test_metaschema_ref.js && node tests/test_pure_js_unsupported.js && node tests/test_native_load_order.js && node tests/test_pack_purity.js && node tests/test_make_native_package.js && node tests/test_browser_nofs.js && node tests/test_browser_imports_guard.js && node tests/test_esm_exports.js && node tests/test_version_sync.js && node tests/test_native_loaded.js && node tests/test_native_lazy.js && node tests/test_cold_start_modules.js && node tests/test_safe_regex_source_sync.js && node tests/test_t_builder.js && node tests/test_async_refine.js && node tests/test_safe_regex.js && node tests/test_safe_regex_integration.js && node tests/test_aot_build.js && node tests/test_aot_differential.js && node tests/test_aot_cli_build.js && node tests/test_aot_cli_smoke.js && node tests/test_bundle_standalone.js && node tests/test_standalone_anyof.js && node tests/test_standalone_formats.js && node tests/test_aot_format_mode.js && node tests/test_aot_additional_props_errors.js && node tests/test_id_anchor_refs.js && node tests/test_engine_routing.js && node tests/test_engine_option.js && node tests/test_compile_cache_order.js && node tests/test_engine_diagnostic.js && node tests/test_format_engine_parity.js && node tests/test_error_accumulator_collision.js && node tests/test_uri_helper_parity.js && node tests/test_uri_fast_path.js && node tests/test_extend_verdict.js && node tests/test_ipv4_regex.js && node tests/test_ipv6_fast_path.js && node tests/test_nested_recursion_decline.js && node tests/test_generator_error_content.js && node tests/test_inherited_key_presence.js && node tests/test_proto_accessor.js && node tests/test_hybrid_tier.js && node tests/test_first_rejection_builds_one.js && node tests/test_container_local.js && node tests/test_one_shot_validate.js && node tests/test_fixed_pattern_inline.js && node tests/test_plan_source.js && node tests/test_fused_remove_additional.js && node tests/test_runtime_parse.js && node tests/test_formats_single_pass.js && node tests/test_email_format_cost.js && node tests/test_format_single_error.js && node tests/test_defs_pointer_alias.js && node tests/test_cross_doc_root_ref.js && node tests/test_codegen_entrypoint_agreement.js && node tests/test_ref_annotation_siblings.js && node tests/test_gate_wrapped_schemas.js && node tests/test_pattern_message_escape.js && node tests/test_nested_defaults_engines.js && node tests/test_nested_coercion_parity.js && node tests/test_preprocess_cache.js && node tests/test_large_schema_codegen.js && node tests/test_base_uri.js && node tests/test_hoisted_def_recursion.js && node tests/test_hybrid_agreement.js && node tests/test_codegen_edge_shapes.js && node tests/test_deferred_def_scope.js && node tests/test_node_scoped_ctx_flags.js && node tests/test_fast_slot_exhaustion.js && node tests/test_ajv_errors.js && node tests/test_custom_keywords.js && node tests/test_strict_schema.js && node tests/test_aot_external_checks.js && node tests/test_ajv_parity.js && node tests/test_user_format_error_path.js && node tests/test_unevaluated_error_path.js && node tests/test_unevaluated_error_shape.js && node tests/test_pattern_properties_errors.js && node tests/test_additional_properties_scaling.js && node tests/test_additional_props_combined_cost.js && node tests/test_no_input_mutation.js && node tests/test_typed_validator_runner.js && node tests/test_define_schema.js && node tests/test_error_codes_lock.js && node tests/test_error_code_lookup.js && node tests/test_error_order.js && node tests/test_error_order_ordinal.js && node tests/test_ref_error_order.js && node tests/test_rejection_shape.js && node tests/test_lazy_normalization.js && node tests/test_value_equality.js && node tests/test_vocabulary.js && node tests/test_schema_scan.js && node tests/test_lazy_errors.js && node tests/test_single_pass_errors.js && node tests/test_lazy_json_errors.js && node tests/test_lazy_instance.js && node tests/test_cyclic_input.js && node tests/test_native_error_codes.js && node tests/test_verdict_preprocess.js && node tests/test_lite_parity.js && node tests/test_lite_bundle_size.js && node tests/test_defaults_own_keys.js && node tests/test_plan_compiler.js && node tests/test_nullable.js && node tests/test_validate_and_parse.js && node tests/test_validate_data.js && node tests/test_enrich_error.js && node tests/test_enrich_received.js && node tests/test_rich_errors_optout.js && node tests/test_error_messages.js && node tests/test_source_positions.js && node tests/fuzz_positions.js && node tests/test_malformed_json_termination.js && node tests/test_data_positions.js && node tests/test_targeted_positions.js && node tests/test_position_map_cost.js && node tests/test_targeted_positions_cost.js && node tests/test_aot_staleness.js && node tests/test_file_positions.js && node tests/test_aot_positions_cost.js && node tests/test_schema_hash.js && node tests/test_render_shared.js && node tests/test_renderers.js && node tests/test_additive_fields.js && node tests/test_diagnostic_source.js && node tests/test_diagnose.js && node tests/test_correlate.js && node tests/test_diagnostics_score.js && node tests/test_runtime_error_dx.js && node tests/test_aot_error_dx.js && node tests/test_abort_early.js && node tests/test_branch_collapse.js && node tests/test_suggestions.js && node tests/test_cli_validate.js && node tests/test_cli_version.js && node benchmark/bench_aot_size.mjs",
62
+ "test": "node test.js && node tests/test_removed_aot_methods.js && node tests/test_no_native.js && node tests/test_no_eval.js && node tests/test_property_dependencies.js && node tests/test_v1_dialect.js && node tests/test_buffer_path_parity.js && node tests/test_text_path_errors.js && node tests/test_compiled_parity.js && node tests/test_buffer_gate.js && node tests/test_scanner_differential.js && node tests/test_validate_json_scanner_cost.js && node tests/test_buffer_reject_cost.js && node tests/test_nan_verdict.js && node tests/test_remove_additional_nested.js && node tests/test_exclusive_bounds.js && node tests/test_engine_differential.js && node tests/test_error_shape_differential.js && node tests/test_output_format.js && node tests/test_standard_json_schema.js && node tests/test_retry_message.js && node tests/test_describe_schema.js && node tests/test_aot_parse.js && node tests/test_draft7_semantics.js && node tests/test_metaschema_ref.js && node tests/test_pure_js_unsupported.js && node tests/test_native_load_order.js && node tests/test_pack_purity.js && node tests/test_make_native_package.js && node tests/test_browser_nofs.js && node tests/test_browser_imports_guard.js && node tests/test_esm_exports.js && node tests/test_version_sync.js && node tests/test_native_loaded.js && node tests/test_native_lazy.js && node tests/test_cold_start_modules.js && node tests/test_safe_regex_source_sync.js && node tests/test_t_builder.js && node tests/test_async_refine.js && node tests/test_safe_regex.js && node tests/test_safe_regex_integration.js && node tests/test_regex_ecma_whitespace.js && node tests/test_aot_build.js && node tests/test_aot_differential.js && node tests/test_aot_cli_build.js && node tests/test_aot_cli_smoke.js && node tests/test_bundle_standalone.js && node tests/test_standalone_anyof.js && node tests/test_standalone_formats.js && node tests/test_aot_format_mode.js && node tests/test_aot_additional_props_errors.js && node tests/test_id_anchor_refs.js && node tests/test_engine_routing.js && node tests/test_engine_option.js && node tests/test_compile_cache_order.js && node tests/test_engine_diagnostic.js && node tests/test_format_engine_parity.js && node tests/test_error_accumulator_collision.js && node tests/test_uri_helper_parity.js && node tests/test_uri_fast_path.js && node tests/test_extend_verdict.js && node tests/test_ipv4_regex.js && node tests/test_ipv6_fast_path.js && node tests/test_nested_recursion_decline.js && node tests/test_generator_error_content.js && node tests/test_inherited_key_presence.js && node tests/test_proto_accessor.js && node tests/test_hybrid_tier.js && node tests/test_first_rejection_builds_one.js && node tests/test_container_local.js && node tests/test_one_shot_validate.js && node tests/test_fixed_pattern_inline.js && node tests/test_plan_source.js && node tests/test_fused_remove_additional.js && node tests/test_runtime_parse.js && node tests/test_formats_single_pass.js && node tests/test_email_format_cost.js && node tests/test_format_single_error.js && node tests/test_defs_pointer_alias.js && node tests/test_cross_doc_root_ref.js && node tests/test_codegen_entrypoint_agreement.js && node tests/test_ref_annotation_siblings.js && node tests/test_gate_wrapped_schemas.js && node tests/test_pattern_message_escape.js && node tests/test_nested_defaults_engines.js && node tests/test_nested_coercion_parity.js && node tests/test_preprocess_cache.js && node tests/test_large_schema_codegen.js && node tests/test_base_uri.js && node tests/test_hoisted_def_recursion.js && node tests/test_alias_definitions.js && node tests/test_property_names_ref.js && node tests/test_hybrid_agreement.js && node tests/test_codegen_edge_shapes.js && node tests/test_deferred_def_scope.js && node tests/test_node_scoped_ctx_flags.js && node tests/test_fast_slot_exhaustion.js && node tests/test_ajv_errors.js && node tests/test_custom_keywords.js && node tests/test_strict_schema.js && node tests/test_aot_external_checks.js && node tests/test_ajv_parity.js && node tests/test_user_format_error_path.js && node tests/test_unevaluated_error_path.js && node tests/test_unevaluated_error_shape.js && node tests/test_pattern_properties_errors.js && node tests/test_additional_properties_scaling.js && node tests/test_additional_props_combined_cost.js && node tests/test_no_input_mutation.js && node tests/test_typed_validator_runner.js && node tests/test_define_schema.js && node tests/test_error_codes_lock.js && node tests/test_error_code_lookup.js && node tests/test_error_order.js && node tests/test_error_order_ordinal.js && node tests/test_ref_error_order.js && node tests/test_rejection_shape.js && node tests/test_lazy_normalization.js && node tests/test_value_equality.js && node tests/test_vocabulary.js && node tests/test_schema_scan.js && node tests/test_lazy_errors.js && node tests/test_single_pass_errors.js && node tests/test_lazy_json_errors.js && node tests/test_lazy_instance.js && node tests/test_cyclic_input.js && node tests/test_native_error_codes.js && node tests/test_verdict_preprocess.js && node tests/test_lite_parity.js && node tests/test_lite_bundle_size.js && node tests/test_defaults_own_keys.js && node tests/test_plan_compiler.js && node tests/test_nullable.js && node tests/test_validate_and_parse.js && node tests/test_validate_data.js && node tests/test_enrich_error.js && node tests/test_enrich_received.js && node tests/test_rich_errors_optout.js && node tests/test_error_messages.js && node tests/test_source_positions.js && node tests/fuzz_positions.js && node tests/test_malformed_json_termination.js && node tests/test_data_positions.js && node tests/test_targeted_positions.js && node tests/test_position_map_cost.js && node tests/test_targeted_positions_cost.js && node tests/test_aot_staleness.js && node tests/test_file_positions.js && node tests/test_aot_positions_cost.js && node tests/test_schema_hash.js && node tests/test_render_shared.js && node tests/test_renderers.js && node tests/test_additive_fields.js && node tests/test_diagnostic_source.js && node tests/test_diagnose.js && node tests/test_correlate.js && node tests/test_diagnostics_score.js && node tests/test_runtime_error_dx.js && node tests/test_aot_error_dx.js && node tests/test_abort_early.js && node tests/test_branch_collapse.js && node tests/test_suggestions.js && node tests/test_cli_validate.js && node tests/test_cli_version.js && node benchmark/bench_aot_size.mjs",
63
63
  "bench:size": "node benchmark/bench_aot_size.mjs",
64
64
  "test:suite": "node tests/run_suite.js && node tests/run_suite.js draft7 && node tests/run_suite.js v1",
65
65
  "test:compat": "node tests/test_compat.js",
@@ -117,6 +117,7 @@
117
117
  "aot.mjs",
118
118
  "aot.d.ts",
119
119
  "lib/",
120
+ "!lib/plan-source.js",
120
121
  "compat.js",
121
122
  "compat.mjs",
122
123
  "compat.d.ts",
@@ -134,13 +135,13 @@
134
135
  "LICENSE"
135
136
  ],
136
137
  "optionalDependencies": {
137
- "@ata-validator/native-darwin-arm64": "1.36.1",
138
- "@ata-validator/native-darwin-x64": "1.36.1",
139
- "@ata-validator/native-linux-arm64-gnu": "1.36.1",
140
- "@ata-validator/native-linux-arm64-musl": "1.36.1",
141
- "@ata-validator/native-linux-x64-gnu": "1.36.1",
142
- "@ata-validator/native-linux-x64-musl": "1.36.1",
143
- "@ata-validator/native-win32-x64": "1.36.1"
138
+ "@ata-validator/native-darwin-arm64": "1.37.1",
139
+ "@ata-validator/native-darwin-x64": "1.37.1",
140
+ "@ata-validator/native-linux-arm64-gnu": "1.37.1",
141
+ "@ata-validator/native-linux-arm64-musl": "1.37.1",
142
+ "@ata-validator/native-linux-x64-gnu": "1.37.1",
143
+ "@ata-validator/native-linux-x64-musl": "1.37.1",
144
+ "@ata-validator/native-win32-x64": "1.37.1"
144
145
  },
145
146
  "peerDependencies": {
146
147
  "yaml": "^2.0.0"
@@ -1,281 +0,0 @@
1
- 'use strict';
2
-
3
- // Generated source from interpreter plans: the verdict function.
4
- //
5
- // The interpreted engine is the reference every other engine is tested
6
- // against, and its Plan is a schema node with every keyword resolved once.
7
- // lib/plan-compiler.js turns plans into closures; this turns them into source,
8
- // deciding at compile time exactly what plan-compiler decides, and calling the
9
- // same helpers (deepEqual, the code point bounds, multipleOfOk, the format
10
- // functions and compiled patterns the plan already holds), so the answers
11
- // cannot drift the way three hand-written generators did.
12
- //
13
- // The scope is deliberately narrow for now: value-level keywords, properties
14
- // and additionalProperties, prefixItems and items, and the in-place
15
- // applicators. Everything else, references, unevaluated*, contains,
16
- // propertyNames, patternProperties, dependentSchemas, custom keywords, makes
17
- // the compiler return null. It never emits a function with a keyword missing.
18
-
19
- const { createInterpreter, _planInternals: I } = require('./interpreter');
20
- const { Plan, deepEqual, cpAtLeast, cpAtMost, multipleOfOk, T_STRING, T_NUMBER, T_INTEGER, T_BOOLEAN, T_NULL, T_OBJECT, T_ARRAY } = I;
21
-
22
- const DECLINE = Symbol('plan-source.decline');
23
- const ALL_TYPES = T_STRING | T_NUMBER | T_INTEGER | T_BOOLEAN | T_NULL | T_OBJECT | T_ARRAY;
24
-
25
- // The own-property test the generators use; see ownKeyExpr in js-compiler.js.
26
- const HELPERS = 'const _hop=Object.prototype.hasOwnProperty;';
27
-
28
- const PROTO_NAMES = new Set([...Object.getOwnPropertyNames(Object.prototype), '__proto__']);
29
-
30
- function isPrimitive(x) {
31
- return x === null || typeof x === 'string' || typeof x === 'boolean' || Number.isFinite(x);
32
- }
33
-
34
- function lit(x) {
35
- return JSON.stringify(x);
36
- }
37
-
38
- function createCtx() {
39
- return { n: 0, argNames: [], argVals: [], fns: [], fnFor: new Map() };
40
- }
41
-
42
- // A value the emitted code needs by reference: a function, a RegExp, a list.
43
- function arg(ctx, value) {
44
- const i = ctx.argVals.indexOf(value);
45
- if (i !== -1) return ctx.argNames[i];
46
- const name = '_a' + ctx.argVals.length;
47
- ctx.argNames.push(name);
48
- ctx.argVals.push(value);
49
- return name;
50
- }
51
-
52
- function local(ctx) {
53
- return '_v' + ctx.n++;
54
- }
55
-
56
- // JSON Schema type bits, exactly as dataBits() assigns them: a non-finite
57
- // number has no type, so it matches neither number nor integer.
58
- function typeTest(mask, v) {
59
- const parts = [];
60
- if (mask & T_STRING) parts.push(`typeof ${v}==='string'`);
61
- if (mask & T_NUMBER) parts.push(`Number.isFinite(${v})`);
62
- else if (mask & T_INTEGER) parts.push(`Number.isInteger(${v})`);
63
- if (mask & T_BOOLEAN) parts.push(`typeof ${v}==='boolean'`);
64
- if (mask & T_NULL) parts.push(`${v}===null`);
65
- if ((mask & T_OBJECT) && (mask & T_ARRAY)) parts.push(`(typeof ${v}==='object'&&${v}!==null)`);
66
- else if (mask & T_OBJECT) parts.push(`(typeof ${v}==='object'&&${v}!==null&&!Array.isArray(${v}))`);
67
- else if (mask & T_ARRAY) parts.push(`Array.isArray(${v})`);
68
- return parts.length ? parts.join('||') : 'false';
69
- }
70
-
71
- // `plain` names a local holding `v.__proto__===Object.prototype`, read once per
72
- // object. With the key a constant at the `in`, the engine keeps the check
73
- // monomorphic; through a shared helper the key is a parameter and it is not.
74
- function ownTest(v, key, plain) {
75
- const k = lit(key);
76
- return PROTO_NAMES.has(key) ? `_hop.call(${v},${k})` : `(${plain}?${k} in ${v}:_hop.call(${v},${k}))`;
77
- }
78
-
79
- const K_NUM = T_NUMBER | T_INTEGER;
80
- // The body under a guard for one kind of value, given what the type check
81
- // already established: unconditional when the value can only be that kind,
82
- // nothing at all when it cannot be.
83
- function guarded(known, kind, cond, body) {
84
- if (!body) return '';
85
- if ((known & kind) === 0) return '';
86
- if ((known & ~kind) === 0) return body;
87
- return `if(${cond}){${body}}`;
88
- }
89
-
90
- // Value-level keywords, in compileLeafV's order and with its meaning.
91
- function emitLeaf(ctx, P, v, known) {
92
- let out = '';
93
- if (P.hasType) out += `if(!(${typeTest(P.typeMask, v)}))return false;`;
94
- if (P.enum !== null) {
95
- const vals = P.enum;
96
- if (vals.length === 0) out += 'return false;';
97
- else if (vals.every(isPrimitive)) out += `if(!(${vals.map((x) => `${v}===${lit(x)}`).join('||')}))return false;`;
98
- else {
99
- const a = arg(ctx, vals);
100
- const de = arg(ctx, deepEqual);
101
- const i = local(ctx);
102
- out += `{let ${i}=0;for(;${i}<${a}.length;${i}++)if(${de}(${a}[${i}],${v}))break;if(${i}===${a}.length)return false}`;
103
- }
104
- }
105
- if (P.hasConst) {
106
- out += isPrimitive(P.const)
107
- ? `if(${v}!==${lit(P.const)})return false;`
108
- : `if(!${arg(ctx, deepEqual)}(${arg(ctx, P.const)},${v}))return false;`;
109
- }
110
- if (P.hasNumber) {
111
- let n = '';
112
- if (P.minimum !== undefined) n += `if(${v}<${lit(P.minimum)})return false;`;
113
- if (P.maximum !== undefined) n += `if(${v}>${lit(P.maximum)})return false;`;
114
- if (P.exclusiveMinimum !== undefined) n += `if(${v}<=${lit(P.exclusiveMinimum)})return false;`;
115
- if (P.exclusiveMaximum !== undefined) n += `if(${v}>=${lit(P.exclusiveMaximum)})return false;`;
116
- if (P.multipleOf !== undefined) n += `if(!${arg(ctx, multipleOfOk)}(${v},${lit(P.multipleOf)}))return false;`;
117
- out += guarded(known, K_NUM, `Number.isFinite(${v})`, n);
118
- }
119
- if (P.hasString) {
120
- let s = '';
121
- if (P.minLength !== undefined) s += `if(!${arg(ctx, cpAtLeast)}(${v},${lit(P.minLength)}))return false;`;
122
- if (P.maxLength !== undefined) s += `if(!${arg(ctx, cpAtMost)}(${v},${lit(P.maxLength)}))return false;`;
123
- if (P.pattern !== null) s += `if(!${arg(ctx, P.pattern)}.test(${v}))return false;`;
124
- if (P.formatFn !== null) s += `if(!${arg(ctx, P.formatFn)}(${v}))return false;`;
125
- out += guarded(known, T_STRING, `typeof ${v}==='string'`, s);
126
- }
127
- if (P.minItems !== undefined || P.maxItems !== undefined || P.uniqueItems) {
128
- let a = '';
129
- if (P.minItems !== undefined) a += `if(${v}.length<${lit(P.minItems)})return false;`;
130
- if (P.maxItems !== undefined) a += `if(${v}.length>${lit(P.maxItems)})return false;`;
131
- if (P.uniqueItems) {
132
- const de = arg(ctx, deepEqual);
133
- const i = local(ctx), j = local(ctx);
134
- a += `for(let ${i}=0;${i}<${v}.length;${i}++)for(let ${j}=${i}+1;${j}<${v}.length;${j}++)if(${de}(${v}[${i}],${v}[${j}]))return false;`;
135
- }
136
- out += guarded(known, T_ARRAY, `Array.isArray(${v})`, a);
137
- }
138
- return out;
139
- }
140
-
141
- // The object-level keywords of a node, value-level and structural together, so
142
- // the prototype of the object is read once for all of them.
143
- function emitObjectLeaf(ctx, P, v, plain) {
144
- let o = '';
145
- if (P.required !== null || P.minProperties !== undefined || P.maxProperties !== undefined || P.dependentRequired !== null) {
146
- if (P.required !== null) for (const k of P.required) o += `if(!${ownTest(v, k, plain)})return false;`;
147
- if (P.minProperties !== undefined) o += `if(Object.keys(${v}).length<${lit(P.minProperties)})return false;`;
148
- if (P.maxProperties !== undefined) o += `if(Object.keys(${v}).length>${lit(P.maxProperties)})return false;`;
149
- if (P.dependentRequired !== null) {
150
- for (const [key, deps] of P.dependentRequired) {
151
- o += `if(${ownTest(v, key, plain)}){${deps.map((dep) => `if(!${ownTest(v, dep, plain)})return false;`).join('')}}`;
152
- }
153
- }
154
- }
155
- return o;
156
- }
157
-
158
- // A subschema asked for a verdict of its own (a branch of anyOf/oneOf, the
159
- // subject of not/if) becomes a named function, emitted once per plan.
160
- function branchFn(ctx, node) {
161
- if (node === true) return '_t';
162
- if (node === false) return '_f';
163
- let name = ctx.fnFor.get(node);
164
- if (name !== undefined) return name;
165
- name = '_b' + ctx.fnFor.size;
166
- ctx.fnFor.set(node, name);
167
- const body = emitNode(ctx, node, 'd');
168
- ctx.fns.push(`function ${name}(d){${body}return true}`);
169
- return name;
170
- }
171
-
172
- function emitNode(ctx, node, v) {
173
- if (node === true) return '';
174
- if (node === false) return 'return false;';
175
- if (!(node instanceof Plan)) return '';
176
- const P = node;
177
- if (P.tracked || P.hasUnevaluated || P.hasCustom || P.macros !== null ||
178
- P.contains !== undefined || P.propertyNames !== undefined || P.patternProperties !== null ||
179
- P.dependentSchemas !== null || P.propertyDependencies !== null) {
180
- throw DECLINE;
181
- }
182
- const known = P.hasType ? P.typeMask : ALL_TYPES;
183
- let out = emitLeaf(ctx, P, v, known);
184
-
185
- if (P.prefixItems !== null || P.items !== undefined) {
186
- let a = '';
187
- if (P.prefixItems !== null) {
188
- P.prefixItems.forEach((child, i) => {
189
- const x = local(ctx);
190
- const body = emitNode(ctx, child, x);
191
- if (body) a += `if(${v}.length>${i}){const ${x}=${v}[${i}];${body}}`;
192
- });
193
- }
194
- if (P.items !== undefined) {
195
- const start = P.prefixItems !== null ? P.prefixItems.length : 0;
196
- const i = local(ctx);
197
- const x = local(ctx);
198
- const body = emitNode(ctx, P.items, x);
199
- if (body) a += `for(let ${i}=${start};${i}<${v}.length;${i}++){const ${x}=${v}[${i}];${body}}`;
200
- }
201
- out += guarded(known, T_ARRAY, `Array.isArray(${v})`, a);
202
- }
203
-
204
- const plain = local(ctx);
205
- let o = emitObjectLeaf(ctx, P, v, plain);
206
- if (P.properties !== null || P.additionalProperties !== undefined) {
207
- const declared = P.properties !== null ? [...P.properties.keys()] : [];
208
- if (P.properties !== null) {
209
- for (const [key, entry] of P.properties) {
210
- const x = local(ctx);
211
- const body = emitNode(ctx, entry.node, x);
212
- if (body) o += `if(${ownTest(v, key, plain)}){const ${x}=${v}[${lit(key)}];${body}}`;
213
- }
214
- }
215
- const ap = P.additionalProperties;
216
- if (ap !== undefined && ap !== true) {
217
- const k = local(ctx);
218
- let isDeclared = 'false';
219
- if (declared.length > 8) {
220
- const set = Object.create(null);
221
- for (const key of declared) set[key] = 1;
222
- isDeclared = `${arg(ctx, set)}[${k}]===1`;
223
- } else if (declared.length > 0) {
224
- isDeclared = declared.map((key) => `${k}===${lit(key)}`).join('||');
225
- }
226
- let apBody;
227
- if (ap === false) apBody = 'return false;';
228
- else {
229
- const x = local(ctx);
230
- const body = emitNode(ctx, ap, x);
231
- apBody = body ? `const ${x}=${v}[${k}];${body}` : '';
232
- }
233
- if (apBody) o += `for(const ${k} in ${v}){if(!${plain}&&!_hop.call(${v},${k}))continue;if(${isDeclared})continue;${apBody}}`;
234
- }
235
- }
236
- if (o) out += guarded(known, T_OBJECT, `typeof ${v}==='object'&&${v}!==null&&!Array.isArray(${v})`, `const ${plain}=${v}.__proto__===Object.prototype;${o}`);
237
-
238
- if (P.allOf !== null) for (const child of P.allOf) out += emitNode(ctx, child, v);
239
- if (P.anyOf !== null) out += `if(!(${P.anyOf.map((c) => `${branchFn(ctx, c)}(${v})`).join('||')}))return false;`;
240
- if (P.oneOf !== null) {
241
- const c = local(ctx);
242
- out += `{let ${c}=0;${P.oneOf.map((b) => `if(${branchFn(ctx, b)}(${v})&&++${c}>1)return false;`).join('')}if(${c}!==1)return false}`;
243
- }
244
- if (P.not !== undefined) out += `if(${branchFn(ctx, P.not)}(${v}))return false;`;
245
- if (P.if !== undefined) {
246
- const t = P.then !== undefined ? emitNode(ctx, P.then, v) : '';
247
- const e = P.else !== undefined ? emitNode(ctx, P.else, v) : '';
248
- if (t || e) out += `if(${branchFn(ctx, P.if)}(${v})){${t}}else{${e}}`;
249
- }
250
- return out;
251
- }
252
-
253
- // Returns { fn, source } or null when the schema is outside what this emits.
254
- function compileVerdict(schema, options) {
255
- let interp;
256
- try {
257
- interp = createInterpreter(schema, options || {});
258
- } catch {
259
- return null;
260
- }
261
- const ctx = createCtx();
262
- let body;
263
- try {
264
- body = emitNode(ctx, interp.rootNode, 'd');
265
- } catch (e) {
266
- if (e === DECLINE) return null;
267
- throw e;
268
- }
269
- const source = HELPERS + 'function _t(){return true}function _f(){return false}' +
270
- ctx.fns.join('') + `return function(d){${body}return true}`;
271
- let fn;
272
- try {
273
- // eslint-disable-next-line no-new-func
274
- fn = new Function(...ctx.argNames, source)(...ctx.argVals);
275
- } catch {
276
- return null;
277
- }
278
- return { fn, source };
279
- }
280
-
281
- module.exports = { compileVerdict };