ata-validator 1.21.0 → 1.22.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +42 -0
- package/README.md +94 -12
- package/compat.js +12 -1
- package/index.d.ts +31 -0
- package/index.js +46 -17
- package/lib/data-positions.js +10 -0
- package/lib/enrich-error.js +89 -26
- package/lib/formats.js +158 -30
- package/lib/interpreter.js +20 -1
- package/lib/js-compiler.js +53 -2
- package/lib/levenshtein.js +19 -7
- package/lib/plan-compiler.js +38 -0
- package/lib/pointer.js +46 -0
- package/lib/retry-message.js +26 -2
- package/lib/suggestions.js +27 -24
- package/lib/version.js +1 -1
- package/package.json +10 -10
package/lib/formats.js
CHANGED
|
@@ -224,24 +224,102 @@ function hostname (s) {
|
|
|
224
224
|
return labelLength !== 0 && previous !== 45;
|
|
225
225
|
}
|
|
226
226
|
|
|
227
|
-
//
|
|
228
|
-
//
|
|
229
|
-
//
|
|
227
|
+
// Character classes as tables rather than comparison chains. Nine `===` tests
|
|
228
|
+
// per character more than doubled the cost of the scan they guarded, measured
|
|
229
|
+
// against the same loop doing a single range check; an indexed byte read costs
|
|
230
|
+
// the same whatever the class holds.
|
|
231
|
+
const URI_CHAR = new Uint8Array(128);
|
|
232
|
+
for (let i = 33; i < 127; i++) URI_CHAR[i] = 1;
|
|
233
|
+
// " < > \ ^ ` { | } are printable and still not allowed in a URI.
|
|
234
|
+
for (const c of [34, 60, 62, 92, 94, 96, 123, 124, 125]) URI_CHAR[c] = 0;
|
|
235
|
+
const HEX_CHAR = new Uint8Array(128);
|
|
236
|
+
for (let i = 48; i < 58; i++) HEX_CHAR[i] = 1;
|
|
237
|
+
for (let i = 97; i < 103; i++) HEX_CHAR[i] = 1;
|
|
238
|
+
for (let i = 65; i < 71; i++) HEX_CHAR[i] = 1;
|
|
239
|
+
// Scheme characters: letters, digits, "+", "-", ".".
|
|
240
|
+
const SCHEME_CHAR = new Uint8Array(128);
|
|
241
|
+
for (let i = 48; i < 58; i++) SCHEME_CHAR[i] = 1;
|
|
242
|
+
for (let i = 97; i < 123; i++) SCHEME_CHAR[i] = 1;
|
|
243
|
+
for (let i = 65; i < 91; i++) SCHEME_CHAR[i] = 1;
|
|
244
|
+
SCHEME_CHAR[43] = 1; SCHEME_CHAR[45] = 1; SCHEME_CHAR[46] = 1;
|
|
245
|
+
|
|
246
|
+
// A scheme followed by a colon, an optional authority, and no character a URI
|
|
247
|
+
// cannot hold anywhere after the colon.
|
|
248
|
+
//
|
|
249
|
+
// One walk answers all of it. The authority's landmarks, the last "@", the
|
|
250
|
+
// colons after it and whether a bracket appeared before it, are recorded while
|
|
251
|
+
// the characters are being checked, so the authority is never cut out of the
|
|
252
|
+
// string or scanned again. Reading it out with `slice` and asking the copy for
|
|
253
|
+
// `lastIndexOf` and two regular expressions allocated a string per URI, and a
|
|
254
|
+
// document carrying a handful of URLs paid that per field.
|
|
230
255
|
function uri (s) {
|
|
231
256
|
const n = s.length;
|
|
232
257
|
if (n === 0) return false;
|
|
233
|
-
|
|
234
|
-
|
|
258
|
+
let c = s.charCodeAt(0);
|
|
259
|
+
// A scheme opens with a letter, never a digit or a sign.
|
|
260
|
+
if (c > 127 || SCHEME_CHAR[c] === 0 || (c >= 48 && c <= 57) || c === 43 || c === 45 || c === 46) return false;
|
|
235
261
|
let colon = -1;
|
|
236
262
|
for (let i = 1; i < n; i++) {
|
|
237
|
-
|
|
263
|
+
c = s.charCodeAt(i);
|
|
238
264
|
if (c === 58) { colon = i; break; }
|
|
239
|
-
|
|
240
|
-
if (!(isDigit(c) || (c >= 97 && c <= 122) || (c >= 65 && c <= 90) ||
|
|
241
|
-
c === 43 || c === 45 || c === 46)) return false;
|
|
265
|
+
if (c > 127 || SCHEME_CHAR[c] === 0) return false;
|
|
242
266
|
}
|
|
243
267
|
if (colon === -1) return false;
|
|
244
|
-
|
|
268
|
+
|
|
269
|
+
let i = colon + 1;
|
|
270
|
+
let authEnd = n;
|
|
271
|
+
if (s.charCodeAt(i) === 47 && s.charCodeAt(i + 1) === 47) {
|
|
272
|
+
const authStart = i + 2;
|
|
273
|
+
let at = -1, firstColon = -1, lastColon = -1, sawBracket = 0, bracketBeforeAt = 0;
|
|
274
|
+
let hostStart = authStart;
|
|
275
|
+
authEnd = -1;
|
|
276
|
+
for (i = authStart; i < n; i++) {
|
|
277
|
+
c = s.charCodeAt(i);
|
|
278
|
+
if (c > 127 || URI_CHAR[c] === 0) return false;
|
|
279
|
+
if (c === 47 || c === 63 || c === 35) { authEnd = i; break; }
|
|
280
|
+
if (c === 37) {
|
|
281
|
+
const h1 = s.charCodeAt(i + 1), h2 = s.charCodeAt(i + 2);
|
|
282
|
+
// A "%" at the end of the string reads as NaN, which no comparison
|
|
283
|
+
// admits; `<= 127` is written that way so NaN takes the reject.
|
|
284
|
+
if (!(h1 <= 127) || !(h2 <= 127) || HEX_CHAR[h1] === 0 || HEX_CHAR[h2] === 0) return false;
|
|
285
|
+
i += 2;
|
|
286
|
+
continue;
|
|
287
|
+
}
|
|
288
|
+
// The last "@" ends userinfo, so everything recorded before it belongs
|
|
289
|
+
// to a part that has its own rules and is dropped here.
|
|
290
|
+
if (c === 64) { bracketBeforeAt = sawBracket; at = i; firstColon = -1; lastColon = -1; hostStart = i + 1; }
|
|
291
|
+
else if (c === 58) { if (firstColon === -1) firstColon = i; lastColon = i; }
|
|
292
|
+
else if (c === 91 || c === 93) { sawBracket = 1; }
|
|
293
|
+
}
|
|
294
|
+
if (authEnd === -1) authEnd = n;
|
|
295
|
+
// Brackets belong to the host, so one before the "@" is not userinfo.
|
|
296
|
+
if (at !== -1 && bracketBeforeAt) return false;
|
|
297
|
+
if (s.charCodeAt(hostStart) === 91) { // '['
|
|
298
|
+
let close = -1;
|
|
299
|
+
for (let j = hostStart + 1; j < authEnd; j++) { if (s.charCodeAt(j) === 93) { close = j; break; } }
|
|
300
|
+
if (close === -1) return false;
|
|
301
|
+
if (close + 1 !== authEnd) {
|
|
302
|
+
if (s.charCodeAt(close + 1) !== 58) return false;
|
|
303
|
+
if (!allDigits(s, close + 2, authEnd)) return false;
|
|
304
|
+
}
|
|
305
|
+
} else if (lastColon !== -1) {
|
|
306
|
+
// A host holding more than one colon is an IPv6 address, and those must
|
|
307
|
+
// be bracketed. Without that rule the last group reads as a port number.
|
|
308
|
+
if (firstColon !== lastColon) return false;
|
|
309
|
+
if (!allDigits(s, lastColon + 1, authEnd)) return false;
|
|
310
|
+
}
|
|
311
|
+
i = authEnd;
|
|
312
|
+
}
|
|
313
|
+
for (; i < n; i++) {
|
|
314
|
+
c = s.charCodeAt(i);
|
|
315
|
+
if (c > 127 || URI_CHAR[c] === 0) return false;
|
|
316
|
+
if (c === 37) {
|
|
317
|
+
const h1 = s.charCodeAt(i + 1), h2 = s.charCodeAt(i + 2);
|
|
318
|
+
if (!(h1 <= 127) || !(h2 <= 127) || HEX_CHAR[h1] === 0 || HEX_CHAR[h2] === 0) return false;
|
|
319
|
+
i += 2;
|
|
320
|
+
}
|
|
321
|
+
}
|
|
322
|
+
return true;
|
|
245
323
|
}
|
|
246
324
|
|
|
247
325
|
// The characters a URI cannot hold: C0 controls, space, DEL, and the rest of
|
|
@@ -287,34 +365,68 @@ const uriCharsSource = (v, from) =>
|
|
|
287
365
|
'if(!((_h1>=48&&_h1<=57)||(_h1>=97&&_h1<=102)||(_h1>=65&&_h1<=70))||' +
|
|
288
366
|
'!((_h2>=48&&_h2<=57)||(_h2>=97&&_h2<=102)||(_h2>=65&&_h2<=70)))return false;_ri+=2}}';
|
|
289
367
|
|
|
368
|
+
function allDigits (s, from, to) {
|
|
369
|
+
for (let i = from; i < to; i++) {
|
|
370
|
+
const c = s.charCodeAt(i);
|
|
371
|
+
if (c < 48 || c > 57) return false;
|
|
372
|
+
}
|
|
373
|
+
return true;
|
|
374
|
+
}
|
|
375
|
+
|
|
290
376
|
// The authority sits between "//" and the next "/", "?" or "#". A bracketed
|
|
291
377
|
// host must close, and a port is digits.
|
|
378
|
+
//
|
|
379
|
+
// Everything here reads the original string through index bounds. Cutting the
|
|
380
|
+
// authority out with `slice` and asking it for `lastIndexOf` and two regular
|
|
381
|
+
// expressions allocated a string for every URI validated, and a document
|
|
382
|
+
// carrying a handful of URLs paid that per field; the authority of an ordinary
|
|
383
|
+
// URL is a dozen characters, so walking it twice costs less than copying it
|
|
384
|
+
// once.
|
|
292
385
|
function uriAuthority (s, start) {
|
|
293
386
|
if (s.charCodeAt(start) !== 47 || s.charCodeAt(start + 1) !== 47) return true;
|
|
294
|
-
|
|
295
|
-
|
|
387
|
+
const n = s.length;
|
|
388
|
+
const authStart = start + 2;
|
|
389
|
+
let authEnd = n;
|
|
390
|
+
for (let i = authStart; i < n; i++) {
|
|
296
391
|
const c = s.charCodeAt(i);
|
|
297
|
-
if (c === 47 || c === 63 || c === 35) {
|
|
298
|
-
}
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
392
|
+
if (c === 47 || c === 63 || c === 35) { authEnd = i; break; }
|
|
393
|
+
}
|
|
394
|
+
// The last "@" splits userinfo from the host. Brackets belong to the host,
|
|
395
|
+
// so one before the "@" is not userinfo.
|
|
396
|
+
let at = -1;
|
|
397
|
+
for (let i = authEnd - 1; i >= authStart; i--) {
|
|
398
|
+
if (s.charCodeAt(i) === 64) { at = i; break; }
|
|
399
|
+
}
|
|
400
|
+
if (at !== -1) {
|
|
401
|
+
for (let i = authStart; i < at; i++) {
|
|
402
|
+
const c = s.charCodeAt(i);
|
|
403
|
+
if (c === 91 || c === 93) return false;
|
|
404
|
+
}
|
|
405
|
+
}
|
|
406
|
+
const hpStart = at === -1 ? authStart : at + 1;
|
|
407
|
+
if (s.charCodeAt(hpStart) === 91) { // '['
|
|
408
|
+
let close = -1;
|
|
409
|
+
for (let i = hpStart + 1; i < authEnd; i++) {
|
|
410
|
+
if (s.charCodeAt(i) === 93) { close = i; break; }
|
|
411
|
+
}
|
|
306
412
|
if (close === -1) return false;
|
|
307
|
-
|
|
308
|
-
if (
|
|
309
|
-
|
|
310
|
-
|
|
413
|
+
if (close + 1 === authEnd) return true;
|
|
414
|
+
if (s.charCodeAt(close + 1) !== 58) return false;
|
|
415
|
+
return allDigits(s, close + 2, authEnd);
|
|
416
|
+
}
|
|
417
|
+
let firstColon = -1;
|
|
418
|
+
let lastColon = -1;
|
|
419
|
+
for (let i = hpStart; i < authEnd; i++) {
|
|
420
|
+
if (s.charCodeAt(i) === 58) {
|
|
421
|
+
if (firstColon === -1) firstColon = i;
|
|
422
|
+
lastColon = i;
|
|
423
|
+
}
|
|
311
424
|
}
|
|
312
|
-
|
|
313
|
-
if (colon === -1) return true;
|
|
425
|
+
if (lastColon === -1) return true;
|
|
314
426
|
// A host holding more than one colon is an IPv6 address, and those must be
|
|
315
427
|
// bracketed. Without that rule the last group reads as a port number.
|
|
316
|
-
if (
|
|
317
|
-
return
|
|
428
|
+
if (firstColon !== lastColon) return false;
|
|
429
|
+
return allDigits(s, lastColon + 1, authEnd);
|
|
318
430
|
}
|
|
319
431
|
const uriAuthoritySource = (v, start) =>
|
|
320
432
|
`if(${v}.charCodeAt(${start})===47&&${v}.charCodeAt(${start}+1)===47){` +
|
|
@@ -622,6 +734,22 @@ function timeSource (v, isStr) {
|
|
|
622
734
|
'if(_u!==1439)return false}');
|
|
623
735
|
}
|
|
624
736
|
|
|
737
|
+
// The same walk as `uri` above, as source, for the code generator to hoist
|
|
738
|
+
// once per compiled function and call. Two copies of one algorithm is what
|
|
739
|
+
// this file has always done for every format, so they sit together here and
|
|
740
|
+
// tests/test_uri_helper_parity.js holds them to the same answer over a corpus
|
|
741
|
+
// built from the walk's own boundaries.
|
|
742
|
+
const URI_HELPER_TABLES = 'const _uct=new Uint8Array(128);for(let _i=33;_i<127;_i++)_uct[_i]=1;' +
|
|
743
|
+
'_uct[34]=_uct[60]=_uct[62]=_uct[92]=_uct[94]=_uct[96]=_uct[123]=_uct[124]=_uct[125]=0;' +
|
|
744
|
+
'const _uch=new Uint8Array(128);for(let _i=48;_i<58;_i++)_uch[_i]=1;' +
|
|
745
|
+
'for(let _i=97;_i<103;_i++)_uch[_i]=1;for(let _i=65;_i<71;_i++)_uch[_i]=1;' +
|
|
746
|
+
'const _ucs=new Uint8Array(128);for(let _i=48;_i<58;_i++)_ucs[_i]=1;' +
|
|
747
|
+
'for(let _i=97;_i<123;_i++)_ucs[_i]=1;for(let _i=65;_i<91;_i++)_ucs[_i]=1;' +
|
|
748
|
+
'_ucs[43]=1;_ucs[45]=1;_ucs[46]=1';
|
|
749
|
+
const URI_HELPER_BODY = 'const _n=_s.length;if(_n===0)return false;let _c=_s.charCodeAt(0);if(_c>127||_ucs[_c]===0||(_c>=48&&_c<=57)||_c===43||_c===45||_c===46)return false;let _co=-1;for(let _i=1;_i<_n;_i++){_c=_s.charCodeAt(_i);if(_c===58){_co=_i;break}if(_c>127||_ucs[_c]===0)return false}if(_co===-1)return false;let _i=_co+1;let _ae=_n;if(_s.charCodeAt(_i)===47&&_s.charCodeAt(_i+1)===47){const _as=_i+2;let _at=-1,_fc=-1,_lc=-1,_br=0,_ba=0,_hs=_as;_ae=-1;for(_i=_as;_i<_n;_i++){_c=_s.charCodeAt(_i);if(_c>127||_uct[_c]===0)return false;if(_c===47||_c===63||_c===35){_ae=_i;break}if(_c===37){const _h1=_s.charCodeAt(_i+1),_h2=_s.charCodeAt(_i+2);if(!(_h1<=127)||!(_h2<=127)||_uch[_h1]===0||_uch[_h2]===0)return false;_i+=2;continue}if(_c===64){_ba=_br;_at=_i;_fc=-1;_lc=-1;_hs=_i+1}else if(_c===58){if(_fc===-1)_fc=_i;_lc=_i}else if(_c===91||_c===93){_br=1}}if(_ae===-1)_ae=_n;if(_at!==-1&&_ba)return false;if(_s.charCodeAt(_hs)===91){let _cl=-1;for(let _j=_hs+1;_j<_ae;_j++){if(_s.charCodeAt(_j)===93){_cl=_j;break}}if(_cl===-1)return false;if(_cl+1!==_ae){if(_s.charCodeAt(_cl+1)!==58)return false;for(let _j=_cl+2;_j<_ae;_j++){const _d=_s.charCodeAt(_j);if(_d<48||_d>57)return false}}}else if(_lc!==-1){if(_fc!==_lc)return false;for(let _j=_lc+1;_j<_ae;_j++){const _d=_s.charCodeAt(_j);if(_d<48||_d>57)return false}}_i=_ae}for(;_i<_n;_i++){_c=_s.charCodeAt(_i);if(_c>127||_uct[_c]===0)return false;if(_c===37){const _h1=_s.charCodeAt(_i+1),_h2=_s.charCodeAt(_i+2);if(!(_h1<=127)||!(_h2<=127)||_uch[_h1]===0||_uch[_h2]===0)return false;_i+=2}}return true;';
|
|
750
|
+
const uriHelperSource = (name) =>
|
|
751
|
+
URI_HELPER_TABLES + ';function ' + name + '(_s){' + URI_HELPER_BODY + '}';
|
|
752
|
+
|
|
625
753
|
function uriSource (v, isStr) {
|
|
626
754
|
return guard(v, isStr, `const _n=${v}.length;if(_n===0)return false;` +
|
|
627
755
|
`const _f=${v}.charCodeAt(0);if(!((_f>=97&&_f<=122)||(_f>=65&&_f<=90)))return false;` +
|
|
@@ -814,4 +942,4 @@ function durationSource (v, isStr) {
|
|
|
814
942
|
return isStr ? inner : `if(typeof ${v}==='string'&&${inner.slice(3)}`;
|
|
815
943
|
}
|
|
816
944
|
|
|
817
|
-
module.exports = { date, ipv4, dateTime, ipv6, hostname, uri, uriReference, uuid, time, noReserved, jsonPointer, relativeJsonPointer, uriTemplate, email, duration, uriChars, uriAuthority, uriCharsSource, uriAuthoritySource, iri, iriReference, idnEmail, iriSource, iriReferenceSource, idnEmailSource, dateSource, ipv4Source, dateTimeSource, ipv6Source, hostnameSource, uriSource, uuidSource, timeSource, noReservedSource, jsonPointerSource, relativeJsonPointerSource, uriTemplateSource, emailSource, durationSource };
|
|
945
|
+
module.exports = { uriHelperSource, date, ipv4, dateTime, ipv6, hostname, uri, uriReference, uuid, time, noReserved, jsonPointer, relativeJsonPointer, uriTemplate, email, duration, uriChars, uriAuthority, uriCharsSource, uriAuthoritySource, iri, iriReference, idnEmail, iriSource, iriReferenceSource, idnEmailSource, dateSource, ipv4Source, dateTimeSource, ipv6Source, hostnameSource, uriSource, uuidSource, timeSource, noReservedSource, jsonPointerSource, relativeJsonPointerSource, uriTemplateSource, emailSource, durationSource };
|
package/lib/interpreter.js
CHANGED
|
@@ -461,6 +461,11 @@ class Plan {
|
|
|
461
461
|
|
|
462
462
|
this.unevaluatedProperties = schema.unevaluatedProperties !== undefined ? child(schema.unevaluatedProperties) : undefined;
|
|
463
463
|
this.unevaluatedItems = schema.unevaluatedItems !== undefined ? child(schema.unevaluatedItems) : undefined;
|
|
464
|
+
// A `false` here rejects the member outright, and the error belongs to this
|
|
465
|
+
// keyword on this object, not to a boolean schema that happens to sit
|
|
466
|
+
// under it. Recorded at compile time so the check below is one test.
|
|
467
|
+
this.unevaluatedPropertiesFalse = schema.unevaluatedProperties === false;
|
|
468
|
+
this.unevaluatedItemsFalse = schema.unevaluatedItems === false;
|
|
464
469
|
this.hasUnevaluated = this.unevaluatedProperties !== undefined || this.unevaluatedItems !== undefined;
|
|
465
470
|
|
|
466
471
|
// Custom keywords present on this node, in schema order. Each op holds
|
|
@@ -1183,7 +1188,14 @@ class Interpreter {
|
|
|
1183
1188
|
if (P.unevaluatedProperties !== undefined && bits === T_OBJECT) {
|
|
1184
1189
|
for (const key of Object.keys(data)) {
|
|
1185
1190
|
if (local.props && local.props.has(key)) continue;
|
|
1186
|
-
if (
|
|
1191
|
+
if (P.unevaluatedPropertiesFalse) {
|
|
1192
|
+
// The name is what is wrong, so the error names it and sits on the
|
|
1193
|
+
// object holding it. Descending instead reported a false boolean
|
|
1194
|
+
// schema at the member's own path, which says nothing about which
|
|
1195
|
+
// key was unexpected and left `params` empty.
|
|
1196
|
+
valid = false;
|
|
1197
|
+
if (errors !== NOERRORS) errors.push(err('unevaluatedProperties', 'unevaluatedProperties', instancePath, schemaPath + '/unevaluatedProperties', { unevaluatedProperty: key }, 'must NOT have unevaluated properties'));
|
|
1198
|
+
} else if (!this.eval(P.unevaluatedProperties, data[key], base, dynScope, errors, instancePath + '/' + escapePointer(key), schemaPath + '/unevaluatedProperties', stack, DISCARD)) valid = false;
|
|
1187
1199
|
if (!local.props) local.props = new Set();
|
|
1188
1200
|
local.props.add(key);
|
|
1189
1201
|
}
|
|
@@ -1191,6 +1203,13 @@ class Interpreter {
|
|
|
1191
1203
|
if (P.unevaluatedItems !== undefined && bits === T_ARRAY) {
|
|
1192
1204
|
for (let i = 0; i < data.length; i++) {
|
|
1193
1205
|
if (local.items && local.items.has(i)) continue;
|
|
1206
|
+
if (P.unevaluatedItemsFalse) {
|
|
1207
|
+
// One error for the array, not one per trailing element: what the
|
|
1208
|
+
// array got wrong is its length past the last evaluated position.
|
|
1209
|
+
valid = false;
|
|
1210
|
+
if (errors !== NOERRORS) errors.push(err('unevaluatedItems', 'unevaluatedItems', instancePath, schemaPath + '/unevaluatedItems', { limit: i }, 'must NOT have more than ' + i + ' items'));
|
|
1211
|
+
break;
|
|
1212
|
+
}
|
|
1194
1213
|
if (!this.eval(P.unevaluatedItems, data[i], base, dynScope, errors, instancePath + '/' + i, schemaPath + '/unevaluatedItems', stack, DISCARD)) valid = false;
|
|
1195
1214
|
if (!local.items) local.items = new Set();
|
|
1196
1215
|
local.items.add(i);
|
package/lib/js-compiler.js
CHANGED
|
@@ -35,6 +35,46 @@ function hoistOnce(ctx, key, code) {
|
|
|
35
35
|
else if (ctx.helperCode) ctx.helperCode.push(code)
|
|
36
36
|
}
|
|
37
37
|
|
|
38
|
+
// Above this many declared names, asking whether a key is one of them by
|
|
39
|
+
// testing it against each name in turn costs a comparison per name per key,
|
|
40
|
+
// which is quadratic in the number of properties: a 1000-property schema with
|
|
41
|
+
// `additionalProperties: false` spent 75 us where the same schema without the
|
|
42
|
+
// keyword took 3, and doubling the properties quadrupled that. Below this many
|
|
43
|
+
// the two are the same within noise and the chain allocates nothing, so it
|
|
44
|
+
// stays. Cross-process medians, one variant per process, properties of type
|
|
45
|
+
// string: 3 to 100 names are a tie; 250 names 8.76 us against 5.49, 500 26.29
|
|
46
|
+
// against 12.47, and with no per-property work 2000 names 273 against 49.
|
|
47
|
+
const AP_LOOKUP_MIN = 128
|
|
48
|
+
|
|
49
|
+
// A name set built once per compiled function rather than once per call, as
|
|
50
|
+
// source text so the standalone output carries it the same way. A null
|
|
51
|
+
// prototype means a key called `toString` is not a member by accident, and a
|
|
52
|
+
// plain property read measured faster than `Set#has` at every size tried.
|
|
53
|
+
function emitNameLookup (ctx, names) {
|
|
54
|
+
const id = `_apn${ctx.varCounter++}`
|
|
55
|
+
const list = names.map((n) => JSON.stringify(n)).join(',')
|
|
56
|
+
const decl = `const ${id}=Object.create(null);for(const _apx of [${list}])${id}[_apx]=1`
|
|
57
|
+
if (ctx.preamble) ctx.preamble.push(decl)
|
|
58
|
+
else if (ctx.helperCode) ctx.helperCode.push(decl)
|
|
59
|
+
else return null
|
|
60
|
+
return id
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
// "every own key of `v` is one of `names`", as a `return false` check. A
|
|
64
|
+
// hoisted lookup above the crossover, the comparison chain below it.
|
|
65
|
+
function apMembershipCheck (ctx, names, v) {
|
|
66
|
+
if (names.length >= AP_LOOKUP_MIN) {
|
|
67
|
+
const id = emitNameLookup(ctx, names)
|
|
68
|
+
if (id !== null) {
|
|
69
|
+
// An indexed loop over `Object.keys`, not `for...in`: it allocates the key
|
|
70
|
+
// array but measured faster at every size tried, by about a quarter.
|
|
71
|
+
const i = ctx.varCounter++
|
|
72
|
+
return `var _apk${i}=Object.keys(${v});for(var _api${i}=0;_api${i}<_apk${i}.length;_api${i}++)if(${id}[_apk${i}[_api${i}]]===undefined)return false`
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
return `for(var _k in ${v})if(${names.map((k) => `_k!==${JSON.stringify(k)}`).join('&&')})return false`
|
|
76
|
+
}
|
|
77
|
+
|
|
38
78
|
function emitDeq(ctx) {
|
|
39
79
|
hoistOnce(ctx, '_deqHoisted', DEQ_HELPER)
|
|
40
80
|
return '_deq'
|
|
@@ -2147,7 +2187,7 @@ function genCode(schema, v, lines, ctx, knownType) {
|
|
|
2147
2187
|
? (propCount <= 15
|
|
2148
2188
|
? `var _n=0;for(var _k in ${v})_n++;if(_n!==${propCount})return false`
|
|
2149
2189
|
: `if(Object.keys(${v}).length!==${propCount})return false`)
|
|
2150
|
-
:
|
|
2190
|
+
: apMembershipCheck(ctx, Object.keys(schema.properties), v)
|
|
2151
2191
|
_deferOrInline(ctx, lines, v, isObj ? inner : `if(typeof ${v}==='object'&&${v}!==null&&!Array.isArray(${v})){${inner}}`)
|
|
2152
2192
|
}
|
|
2153
2193
|
|
|
@@ -3037,6 +3077,13 @@ function genCode(schema, v, lines, ctx, knownType) {
|
|
|
3037
3077
|
// the check was inlined at each of them.
|
|
3038
3078
|
const EMAIL_HELPER = `function _em(_s){${_formats.emailSource('_s', true)}return true}`
|
|
3039
3079
|
|
|
3080
|
+
// `uri` is the most expensive format an ordinary document carries: a schema
|
|
3081
|
+
// with a handful of URL fields spent more time here than on every structural
|
|
3082
|
+
// check put together. The walk reads its character classes out of hoisted
|
|
3083
|
+
// tables, so it is declared once per compiled function rather than inlined at
|
|
3084
|
+
// each call site, where the tables would have to be rebuilt.
|
|
3085
|
+
const URI_HELPER = _formats.uriHelperSource('_uri')
|
|
3086
|
+
|
|
3040
3087
|
const FORMAT_CODEGEN = {
|
|
3041
3088
|
email: (v, isStr, ctx) => {
|
|
3042
3089
|
if (!ctx) return _formats.emailSource(v, isStr)
|
|
@@ -3058,7 +3105,11 @@ const FORMAT_CODEGEN = {
|
|
|
3058
3105
|
'date-time': _formats.dateTimeSource,
|
|
3059
3106
|
time: _formats.timeSource,
|
|
3060
3107
|
duration: _formats.durationSource,
|
|
3061
|
-
uri:
|
|
3108
|
+
uri: (v, isStr, ctx) => {
|
|
3109
|
+
if (!ctx) return _formats.uriSource(v, isStr)
|
|
3110
|
+
hoistOnce(ctx, '_uriHoisted', URI_HELPER)
|
|
3111
|
+
return isStr ? `if(!_uri(${v}))return false` : `if(typeof ${v}==='string'&&!_uri(${v}))return false`
|
|
3112
|
+
},
|
|
3062
3113
|
'uri-reference': (v, isStr) => isStr
|
|
3063
3114
|
? `{${_formats.uriCharsSource(v, '0')}}`
|
|
3064
3115
|
: `if(typeof ${v}==='string'){${_formats.uriCharsSource(v, '0')}}`,
|
package/lib/levenshtein.js
CHANGED
|
@@ -21,19 +21,31 @@ function levenshtein (a, b, maxDistance) {
|
|
|
21
21
|
}
|
|
22
22
|
let prev = scratchA;
|
|
23
23
|
let curr = scratchB;
|
|
24
|
-
|
|
24
|
+
const bn = b.length;
|
|
25
|
+
for (let j = 0; j <= bn; j++) prev[j] = j;
|
|
25
26
|
for (let i = 1; i <= a.length; i++) {
|
|
26
27
|
curr[0] = i;
|
|
27
28
|
let rowMin = i;
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
29
|
+
// charCodeAt rather than indexing: reading a character by index allocates
|
|
30
|
+
// a one-character string, and this is the innermost loop.
|
|
31
|
+
const ca = a.charCodeAt(i - 1);
|
|
32
|
+
for (let j = 1; j <= bn; j++) {
|
|
33
|
+
const cost = ca === b.charCodeAt(j - 1) ? 0 : 1;
|
|
34
|
+
let m = prev[j - 1] + cost;
|
|
35
|
+
const del = curr[j - 1] + 1;
|
|
36
|
+
if (del < m) m = del;
|
|
37
|
+
const ins = prev[j] + 1;
|
|
38
|
+
if (ins < m) m = ins;
|
|
39
|
+
curr[j] = m;
|
|
40
|
+
if (m < rowMin) rowMin = m;
|
|
32
41
|
}
|
|
33
42
|
if (rowMin > max) return Infinity;
|
|
34
|
-
|
|
43
|
+
// A destructured swap builds an array per row; a temporary does not.
|
|
44
|
+
const t = prev;
|
|
45
|
+
prev = curr;
|
|
46
|
+
curr = t;
|
|
35
47
|
}
|
|
36
|
-
return prev[
|
|
48
|
+
return prev[bn];
|
|
37
49
|
}
|
|
38
50
|
|
|
39
51
|
module.exports = { levenshtein };
|
package/lib/plan-compiler.js
CHANGED
|
@@ -894,6 +894,27 @@ function install(deps) {
|
|
|
894
894
|
// unevaluated*: run last, against the annotations of everything above.
|
|
895
895
|
if (P.unevaluatedProperties !== undefined) {
|
|
896
896
|
const fn = child(P.unevaluatedProperties);
|
|
897
|
+
if (fn === FALSE_PAIR) {
|
|
898
|
+
// `false` rejects the member outright, and what is wrong is the name.
|
|
899
|
+
// Running the boolean schema against the value instead reported a
|
|
900
|
+
// false schema at the member's own path with empty params, which does
|
|
901
|
+
// not say which key was unexpected. The error belongs to this keyword,
|
|
902
|
+
// on the object holding the key, and it carries the key.
|
|
903
|
+
steps.push((data, errors, instancePath, schemaPath, stack, rec) => {
|
|
904
|
+
if (dataBits(data) !== T_OBJECT) return true;
|
|
905
|
+
let ok = true;
|
|
906
|
+
const keys = keysOf(rec, data);
|
|
907
|
+
for (let k = 0; k < keys.length; k++) {
|
|
908
|
+
const key = keys[k];
|
|
909
|
+
if (hasProp(rec, key)) continue;
|
|
910
|
+
ok = false;
|
|
911
|
+
if (errors === NOERRORS) return false;
|
|
912
|
+
errors.push(err('unevaluatedProperties', 'unevaluatedProperties', instancePath, schemaPath + '/unevaluatedProperties', { unevaluatedProperty: key }, 'must NOT have unevaluated properties'));
|
|
913
|
+
addProp(rec, key);
|
|
914
|
+
}
|
|
915
|
+
return ok;
|
|
916
|
+
});
|
|
917
|
+
} else {
|
|
897
918
|
steps.push((data, errors, instancePath, schemaPath, stack, rec) => {
|
|
898
919
|
if (dataBits(data) !== T_OBJECT) return true;
|
|
899
920
|
let ok = true;
|
|
@@ -909,6 +930,7 @@ function install(deps) {
|
|
|
909
930
|
}
|
|
910
931
|
return ok;
|
|
911
932
|
});
|
|
933
|
+
}
|
|
912
934
|
if (fn === FALSE_PAIR) {
|
|
913
935
|
// Every key must already be evaluated; on success there is nothing
|
|
914
936
|
// new to record.
|
|
@@ -936,6 +958,21 @@ function install(deps) {
|
|
|
936
958
|
}
|
|
937
959
|
if (P.unevaluatedItems !== undefined) {
|
|
938
960
|
const fn = child(P.unevaluatedItems);
|
|
961
|
+
if (fn === FALSE_PAIR) {
|
|
962
|
+
// One error for the array rather than one per trailing element: what
|
|
963
|
+
// the array got wrong is its length past the last evaluated position.
|
|
964
|
+
steps.push((data, errors, instancePath, schemaPath, stack, rec) => {
|
|
965
|
+
if (dataBits(data) !== T_ARRAY) return true;
|
|
966
|
+
for (let i = 0; i < data.length; i++) {
|
|
967
|
+
if (hasItem(rec, i)) continue;
|
|
968
|
+
if (errors !== NOERRORS) errors.push(err('unevaluatedItems', 'unevaluatedItems', instancePath, schemaPath + '/unevaluatedItems', { limit: i }, 'must NOT have more than ' + i + ' items'));
|
|
969
|
+
if (data.length > rec.n) rec.n = data.length;
|
|
970
|
+
return false;
|
|
971
|
+
}
|
|
972
|
+
if (data.length > rec.n) rec.n = data.length;
|
|
973
|
+
return true;
|
|
974
|
+
});
|
|
975
|
+
} else {
|
|
939
976
|
steps.push((data, errors, instancePath, schemaPath, stack, rec) => {
|
|
940
977
|
if (dataBits(data) !== T_ARRAY) return true;
|
|
941
978
|
let ok = true;
|
|
@@ -949,6 +986,7 @@ function install(deps) {
|
|
|
949
986
|
if (data.length > rec.n) rec.n = data.length;
|
|
950
987
|
return ok;
|
|
951
988
|
});
|
|
989
|
+
}
|
|
952
990
|
if (fn === FALSE_PAIR) {
|
|
953
991
|
vsteps.push((data, stack, rec) => {
|
|
954
992
|
if (dataBits(data) !== T_ARRAY) return true;
|
package/lib/pointer.js
ADDED
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
'use strict';
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* Resolve a JSON pointer against a document, for the diagnostic paths that run
|
|
5
|
+
* once per error on a rejected payload.
|
|
6
|
+
*
|
|
7
|
+
* Segments are read straight out of the pointer string: no leading-slash
|
|
8
|
+
* regex, no parts array, and no unescape pass on the segments that carry no
|
|
9
|
+
* `~`. A pointer that leaves the document returns rather than throwing,
|
|
10
|
+
* because a diagnostic must not fail where validation succeeded.
|
|
11
|
+
*
|
|
12
|
+
* `missing` separates the two ways a pointer yields nothing: a key that is
|
|
13
|
+
* absent from a container that exists resolves to undefined, while a pointer
|
|
14
|
+
* that walks through a null or a primitive returns `missing`. Callers that
|
|
15
|
+
* format the value need that apart, since the first is a value worth printing
|
|
16
|
+
* and the second is a path that was never in the document.
|
|
17
|
+
*
|
|
18
|
+
* @param {*} data the document the pointer is read against
|
|
19
|
+
* @param {string} pointer an RFC 6901 pointer, '' for the document itself
|
|
20
|
+
* @param {*} [missing] returned when the walk leaves the document
|
|
21
|
+
* @returns {*} the value at the pointer, `missing` if the walk broke
|
|
22
|
+
*/
|
|
23
|
+
function resolvePointer (data, pointer, missing) {
|
|
24
|
+
if (!pointer) return data;
|
|
25
|
+
const len = pointer.length;
|
|
26
|
+
let cur = data;
|
|
27
|
+
let i = pointer.charCodeAt(0) === 47 ? 1 : 0; // 47 is '/'
|
|
28
|
+
for (;;) {
|
|
29
|
+
let j = pointer.indexOf('/', i);
|
|
30
|
+
if (j === -1) j = len;
|
|
31
|
+
let seg = pointer.slice(i, j);
|
|
32
|
+
if (seg.indexOf('~') !== -1) seg = seg.replace(/~1/g, '/').replace(/~0/g, '~');
|
|
33
|
+
if (cur == null) return missing;
|
|
34
|
+
cur = cur[seg];
|
|
35
|
+
if (j === len) break;
|
|
36
|
+
i = j + 1;
|
|
37
|
+
}
|
|
38
|
+
return cur;
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
// Passed where a caller has no resolved value to offer, which is not the same
|
|
42
|
+
// as resolving to undefined: the first means walk it yourself, the second
|
|
43
|
+
// means the document really holds nothing there.
|
|
44
|
+
const UNRESOLVED = Symbol('ata.pointer.unresolved');
|
|
45
|
+
|
|
46
|
+
module.exports = { resolvePointer, UNRESOLVED };
|
package/lib/retry-message.js
CHANGED
|
@@ -22,15 +22,39 @@
|
|
|
22
22
|
// `received`. The trap is that joining `message`, which is the obvious thing to
|
|
23
23
|
// do, throws both away. So this is one call that does not.
|
|
24
24
|
|
|
25
|
+
// `multipleOf: 0.01` is how a schema says money, and a model acts on "rounded
|
|
26
|
+
// to 2 decimal places" where it does not reliably act on "a multiple of 0.01".
|
|
27
|
+
// describeSchema found that by measurement; the same wording belongs here, or
|
|
28
|
+
// the two halves of the loop describe the same rule differently.
|
|
29
|
+
function phrase (error) {
|
|
30
|
+
const p = error.params || {};
|
|
31
|
+
if (error.keyword === 'multipleOf' && typeof p.multipleOf === 'number') {
|
|
32
|
+
const m = p.multipleOf;
|
|
33
|
+
if (m > 0 && m < 1) {
|
|
34
|
+
const places = Math.round(Math.log10(1 / m));
|
|
35
|
+
if (Math.abs(Math.pow(10, -places) - m) < Number.EPSILON * 8) {
|
|
36
|
+
return `must be rounded to ${places} decimal place${places === 1 ? '' : 's'}`;
|
|
37
|
+
}
|
|
38
|
+
}
|
|
39
|
+
}
|
|
40
|
+
return null;
|
|
41
|
+
}
|
|
42
|
+
|
|
25
43
|
function line (error) {
|
|
26
44
|
const where = error.instancePath || error.path || '';
|
|
27
|
-
const body =
|
|
45
|
+
const body = phrase(error) ||
|
|
46
|
+
(typeof error.detail === 'string' && error.detail ? error.detail : error.message);
|
|
28
47
|
// `detail` usually quotes the offending value already; saying it twice reads
|
|
29
48
|
// like a stutter in a string that is going into a prompt. A container is
|
|
30
49
|
// summarised as `[object, ~0.1KB]` rather than shown, which tells a model
|
|
31
50
|
// nothing it cannot see in its own output, so that is left out too.
|
|
32
51
|
const raw = error.received === undefined || error.received === null ? '' : String(error.received);
|
|
33
|
-
|
|
52
|
+
// A container tells a model nothing its own output does not already show,
|
|
53
|
+
// and it is the most expensive thing you can put in a prompt. That covers
|
|
54
|
+
// both the `[object, ~0.1KB]` summary and a whole serialized object.
|
|
55
|
+
const head = raw.charAt(0);
|
|
56
|
+
const useless = (head === '[' && /^\[(object|array)\b/.test(raw)) || head === '{' ||
|
|
57
|
+
(head === '[' && raw.charAt(raw.length - 1) === ']');
|
|
34
58
|
const received = useless ? '' : raw;
|
|
35
59
|
const got = received && body && !body.includes(received) ? `, got ${received}` : '';
|
|
36
60
|
return `${where || '/'}: ${body}${got}`;
|