ata-validator 1.6.1 → 1.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,132 @@
1
+ 'use strict';
2
+
3
+ // The buffer APIs (isValid, countValid, batchIsValid, isValidNDJSON,
4
+ // isValidParallel, isValidPrepadded, and isValidJSON above the simdjson
5
+ // threshold) answer from the native engine's own walker, which is not the
6
+ // engine validate() uses. Measured over the official suite, the two disagreed
7
+ // on 245 of 2222 cases, and 195 of those were the native walker accepting a
8
+ // document validate() rejects. A buffer API that is faster and sometimes
9
+ // wrong in the accepting direction is not a fast path, it is a hole.
10
+ //
11
+ // This module decides, once per schema, whether the native walker can be
12
+ // trusted for it. When it cannot, the buffer APIs are replaced with versions
13
+ // that parse the bytes and call validate(), so every entry point gives the
14
+ // same verdict. The list below is the set of shapes where the suite showed
15
+ // the walker disagreeing, each with the reason. It is deliberately a list of
16
+ // shapes rather than a count: a new disagreement is a new entry here, with
17
+ // tests/test_buffer_path_parity.js holding the total at zero.
18
+
19
+ // Keywords the native walker does not implement, or implements differently
20
+ // enough to disagree on the suite.
21
+ const UNSUPPORTED_KEYWORDS = new Set([
22
+ 'contains', 'minContains', 'maxContains',
23
+ 'unevaluatedProperties', 'unevaluatedItems',
24
+ 'dependencies', 'dependentSchemas', 'dependentRequired',
25
+ 'propertyNames', 'patternProperties',
26
+ ]);
27
+
28
+ // Formats the native checkers answer differently from the JS ones.
29
+ const DIVERGENT_FORMATS = new Set(['hostname', 'date-time', 'time', 'uri-reference', 'duration']);
30
+
31
+ const SUBSCHEMA_MAPS = ['properties', 'patternProperties', '$defs', 'definitions', 'dependentSchemas', 'dependencies'];
32
+ const SUBSCHEMA_NODES = [
33
+ 'items', 'additionalItems', 'additionalProperties', 'contains', 'propertyNames', 'not',
34
+ 'if', 'then', 'else', 'unevaluatedProperties', 'unevaluatedItems',
35
+ 'allOf', 'anyOf', 'oneOf', 'prefixItems',
36
+ ];
37
+
38
+ function walk(schema, depth) {
39
+ if (schema === true || schema === false) {
40
+ // A boolean root is answered wrong by the walker; nested booleans are
41
+ // handled where they appear (items and prefixItems below).
42
+ return depth === 0;
43
+ }
44
+ if (schema === null || typeof schema !== 'object') return false;
45
+ if (Array.isArray(schema)) {
46
+ for (const s of schema) if (walk(s, depth + 1)) return true;
47
+ return false;
48
+ }
49
+ for (const key of Object.keys(schema)) {
50
+ const v = schema[key];
51
+ if (UNSUPPORTED_KEYWORDS.has(key)) return true;
52
+ // Cross-document references: the native engine has no registry of
53
+ // external schemas, and embedded $id changes the base URI in ways its
54
+ // resolver gets wrong.
55
+ if (key === '$ref' && typeof v === 'string' && !v.startsWith('#')) return true;
56
+ if (key === '$id' && depth > 0) return true;
57
+ // An empty enum rejects everything; the walker accepts everything.
58
+ if (key === 'enum' && Array.isArray(v) && v.length === 0) return true;
59
+ if (key === 'format' && DIVERGENT_FORMATS.has(v)) return true;
60
+ // Unicode property escapes: RE2 cannot parse them and the walker then
61
+ // skips the pattern instead of failing.
62
+ if (key === 'pattern' && typeof v === 'string' && /\\[pP]\{/.test(v)) return true;
63
+ // Tuple forms: prefixItems and the draft-07 array form of items are
64
+ // checked against the wrong positions by the walker.
65
+ if (key === 'prefixItems') return true;
66
+ if ((key === 'items' || key === 'additionalItems') && (typeof v === 'boolean' || Array.isArray(v))) return true;
67
+
68
+ if (SUBSCHEMA_MAPS.includes(key)) {
69
+ if (v && typeof v === 'object' && !Array.isArray(v)) {
70
+ for (const k of Object.keys(v)) if (walk(v[k], depth + 1)) return true;
71
+ }
72
+ } else if (SUBSCHEMA_NODES.includes(key)) {
73
+ if (walk(v, depth + 1)) return true;
74
+ }
75
+ }
76
+ return false;
77
+ }
78
+
79
+ // True when the buffer APIs must answer through validate() for this schema.
80
+ function bufferNeedsSlowPath(schema, schemaMap) {
81
+ if (walk(schema, 0)) return true;
82
+ if (schemaMap && schemaMap.size > 0) {
83
+ for (const s of schemaMap.values()) if (walk(s, 1)) return true;
84
+ }
85
+ return false;
86
+ }
87
+
88
+ function toText(input, name) {
89
+ if (typeof input === 'string') return input;
90
+ if (input instanceof Uint8Array) return Buffer.from(input.buffer, input.byteOffset, input.byteLength).toString('utf8');
91
+ throw new TypeError(`${name}() requires a Buffer, Uint8Array, or string. For parsed objects, use isValidObject().`);
92
+ }
93
+
94
+ // Replaces the instance's buffer APIs with versions that parse and call
95
+ // validate(). Installed after compilation, so `validator.validate` is final.
96
+ function installSlowBufferApis(validator) {
97
+ const isValidText = (text) => {
98
+ let value;
99
+ try { value = JSON.parse(text); } catch { return false; }
100
+ return validator.validate(value).valid;
101
+ };
102
+ validator.isValid = (input) => isValidText(toText(input, 'isValid'));
103
+ validator.isValidJSON = (jsonStr) => isValidText(jsonStr);
104
+ validator.isValidPrepadded = (paddedBuffer, jsonLength) =>
105
+ isValidText(Buffer.from(paddedBuffer.buffer, paddedBuffer.byteOffset, jsonLength).toString('utf8'));
106
+ const ndjson = (input, name) => {
107
+ const lines = toText(input, name).split('\n');
108
+ const out = [];
109
+ for (const line of lines) {
110
+ if (line === '') continue; // the native loop skips zero-length lines only
111
+ out.push(isValidText(line));
112
+ }
113
+ return out;
114
+ };
115
+ validator.isValidNDJSON = (input) => ndjson(input, 'isValidNDJSON');
116
+ validator.isValidParallel = (input) => ndjson(input, 'isValidParallel');
117
+ validator.countValid = (input) => {
118
+ let n = 0;
119
+ for (const ok of ndjson(input, 'countValid')) if (ok) n++;
120
+ return n;
121
+ };
122
+ validator.batchIsValid = (buffers) => {
123
+ let n = 0;
124
+ for (const b of buffers) {
125
+ if (!(b instanceof Uint8Array)) throw new TypeError('batchIsValid() requires Buffer or Uint8Array elements');
126
+ if (isValidText(toText(b, 'batchIsValid'))) n++;
127
+ }
128
+ return n;
129
+ };
130
+ }
131
+
132
+ module.exports = { bufferNeedsSlowPath, installSlowBufferApis };
package/lib/draft7.js CHANGED
@@ -9,15 +9,37 @@ function isDraft7(schema) {
9
9
  return !!(schema && schema.$schema && DRAFT7_SCHEMAS.has(schema.$schema))
10
10
  }
11
11
 
12
- function normalizeDraft7(schema) {
13
- if (!isDraft7(schema)) return schema
12
+ // `force` applies the draft-07 rules to a document that declares no dialect
13
+ // of its own, which is how a retrieved schema inherits the root's draft.
14
+ function normalizeDraft7(schema, force) {
15
+ if (!force && !isDraft7(schema)) return schema
14
16
  _normalize(schema)
15
17
  return schema
16
18
  }
17
19
 
20
+ // Keywords that may sit next to `$ref` in draft-07 without being applied:
21
+ // definitions are addressable by pointer, annotations are inert.
22
+ const REF_SIBLINGS_KEPT = new Set(['$ref', '$defs', 'definitions', '$schema', '$comment', 'title', 'description', 'examples', 'default', 'readOnly', 'writeOnly'])
23
+
18
24
  function _normalize(schema) {
19
25
  if (typeof schema !== 'object' || schema === null) return
20
26
 
27
+ // In draft-07 an object with `$ref` is that reference and nothing else:
28
+ // sibling keywords, `$id` included, are ignored. Dropping them here gives
29
+ // every engine the same reading without each having to know the draft.
30
+ if (typeof schema.$ref === 'string') {
31
+ for (const key of Object.keys(schema)) {
32
+ if (!REF_SIBLINGS_KEPT.has(key)) delete schema[key]
33
+ }
34
+ }
35
+
36
+ // A fragment-only `$id` is a plain-name anchor in draft-07; 2020-12 spells
37
+ // it `$anchor`, which every engine already resolves.
38
+ if (typeof schema.$id === 'string' && /^#[A-Za-z][A-Za-z0-9_.:-]*$/.test(schema.$id)) {
39
+ if (schema.$anchor === undefined) schema.$anchor = schema.$id.slice(1)
40
+ delete schema.$id
41
+ }
42
+
21
43
  // definitions → $defs
22
44
  if (schema.definitions && !schema.$defs) {
23
45
  schema.$defs = schema.definitions
@@ -46,14 +46,24 @@ function expectedFor (err) {
46
46
 
47
47
  function pickReceived (err, data) {
48
48
  if (!data && data !== 0 && data !== false) return undefined;
49
- // Walk JSON pointer to extract the actual offending value
49
+ // Walk JSON pointer to extract the actual offending value. This runs for
50
+ // every error on every rejected payload, so the walk reads segments straight
51
+ // out of the pointer: no leading-slash regex, no parts array, and no unescape
52
+ // pass on the segments that carry no `~`.
50
53
  const p = err.instancePath || err.path || '';
51
54
  if (!p) return reprValue(data);
52
- const parts = p.replace(/^\//, '').split('/').map(s => s.replace(/~1/g, '/').replace(/~0/g, '~'));
55
+ const len = p.length;
53
56
  let cur = data;
54
- for (const part of parts) {
57
+ let i = p.charCodeAt(0) === 47 ? 1 : 0; // 47 is '/'
58
+ for (;;) {
59
+ let j = p.indexOf('/', i);
60
+ if (j === -1) j = len;
61
+ let seg = p.slice(i, j);
62
+ if (seg.indexOf('~') !== -1) seg = seg.replace(/~1/g, '/').replace(/~0/g, '~');
55
63
  if (cur == null) return undefined;
56
- cur = cur[part];
64
+ cur = cur[seg];
65
+ if (j === len) break;
66
+ i = j + 1;
57
67
  }
58
68
  return reprValue(cur);
59
69
  }
@@ -77,20 +77,27 @@ function all () {
77
77
  return Object.keys(CODES).sort();
78
78
  }
79
79
 
80
+ // Reverse lookups, built once. codeFor runs per error on every failing
81
+ // validation, so walking the table there showed up as most of the cost of
82
+ // enriching a rejected payload. Insertion follows sorted code order, so the
83
+ // first writer wins and each keyword keeps the lowest code that carries it.
84
+ const BY_KEYWORD = new Map();
85
+ const BY_FORMAT = new Map();
86
+ for (const c of Object.keys(CODES).sort()) {
87
+ const meta = CODES[c];
88
+ if (!BY_KEYWORD.has(meta.keyword)) BY_KEYWORD.set(meta.keyword, c);
89
+ if (meta.keyword === 'format' && meta.format && !BY_FORMAT.has(meta.format)) BY_FORMAT.set(meta.format, c);
90
+ }
91
+
80
92
  // Map a (keyword, optional format) tuple back to a code. Used by the codegen
81
93
  // integration to attach codes to existing error sites without rewriting them all.
82
94
  function codeFor (keyword, format) {
83
95
  if (keyword === 'format' && format) {
84
- for (const c of all()) {
85
- const meta = CODES[c];
86
- if (meta.keyword === 'format' && meta.format === format) return c;
87
- }
88
- return 'ATA3099';
89
- }
90
- for (const c of all()) {
91
- if (CODES[c].keyword === keyword) return c;
96
+ const hit = BY_FORMAT.get(format);
97
+ return hit === undefined ? 'ATA3099' : hit;
92
98
  }
93
- return null;
99
+ const hit = BY_KEYWORD.get(keyword);
100
+ return hit === undefined ? null : hit;
94
101
  }
95
102
 
96
103
  module.exports = { CODES, get, all, codeFor };