soml-lang 0.0.2 → 0.0.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/distribution/edit.d.ts +56 -0
- package/distribution/edit.js +635 -0
- package/distribution/error.d.ts +19 -2
- package/distribution/error.js +20 -3
- package/distribution/format.d.ts +48 -4
- package/distribution/format.js +209 -49
- package/distribution/index.d.ts +6 -3
- package/distribution/index.js +7 -2
- package/distribution/parse.d.ts +68 -5
- package/distribution/parse.js +593 -177
- package/distribution/shared.d.ts +58 -4
- package/distribution/shared.js +93 -11
- package/distribution/stringify.d.ts +88 -8
- package/distribution/stringify.js +173 -59
- package/distribution/tree.d.ts +115 -4
- package/distribution/tree.js +221 -86
- package/package.json +4 -3
- package/readme.md +254 -47
package/distribution/parse.js
CHANGED
|
@@ -1,43 +1,41 @@
|
|
|
1
1
|
import { ParseError } from "./error.js";
|
|
2
|
-
import {
|
|
3
|
-
|
|
4
|
-
const LF = 0x0A;
|
|
5
|
-
const SPACE = 0x20;
|
|
6
|
-
const DOUBLE_QUOTE = 0x22;
|
|
7
|
-
const HASH = 0x23;
|
|
8
|
-
const SINGLE_QUOTE = 0x27;
|
|
9
|
-
const ASTERISK = 0x2A;
|
|
10
|
-
const COMMA = 0x2C;
|
|
11
|
-
const DASH = 0x2D;
|
|
12
|
-
const DOT = 0x2E;
|
|
13
|
-
const SLASH = 0x2F;
|
|
14
|
-
const COLON = 0x3A;
|
|
15
|
-
const OPEN_BRACKET = 0x5B;
|
|
16
|
-
const CLOSE_BRACKET = 0x5D;
|
|
17
|
-
const OPEN_BRACE = 0x7B;
|
|
18
|
-
const CLOSE_BRACE = 0x7D;
|
|
19
|
-
/*
|
|
20
|
-
What may directly follow a scalar value: whitespace, a separator, a closing bracket, a comment, or the end.
|
|
21
|
-
*/
|
|
2
|
+
import { formatKey } from "./stringify.js";
|
|
3
|
+
import { MAX_DEPTH, INT64_MIN, INT64_MAX, MIN_INSTANT, MAX_INSTANT, DURATION_UNITS, createDuration, requireTemporal, trimTrailingZeros, getIntegersOption, describeCharacter, formatCodePoint, describeKey, isBareKeyCharacter, isSpace, skipSpaces, skipSpacesBack, isBlankLine, findNumberEnd, findLineEnd, findBlockStringEnd, abbreviate, LF, SPACE, DOUBLE_QUOTE, HASH, SINGLE_QUOTE, ASTERISK, COMMA, DASH, DOT, SLASH, COLON, OPEN_BRACKET, CLOSE_BRACKET, OPEN_BRACE, CLOSE_BRACE, } from "./shared.js";
|
|
22
4
|
const valueEndCharacters = new Uint8Array(128);
|
|
23
5
|
for (const character of ' \t\n,]}#/') {
|
|
24
6
|
valueEndCharacters[character.codePointAt(0)] = 1;
|
|
25
7
|
}
|
|
26
|
-
const RADIX_DIGIT = {
|
|
27
|
-
x: code => (code >= 0x30 && code <= 0x39) || (code >= 0x41 && code <= 0x46),
|
|
28
|
-
o: code => code >= 0x30 && code <= 0x37,
|
|
29
|
-
b: code => code === 0x30 || code === 0x31,
|
|
30
|
-
};
|
|
31
8
|
/*
|
|
32
|
-
|
|
9
|
+
Whether a UTF-16 code unit may directly follow a scalar value: whitespace, a separator, a closing bracket, a comment, or the end, where `charCodeAt()` returns `NaN`.
|
|
10
|
+
*/
|
|
11
|
+
function isValueEnd(code) {
|
|
12
|
+
return Number.isNaN(code) || (code < 128 && valueEndCharacters[code] === 1);
|
|
13
|
+
}
|
|
14
|
+
/*
|
|
15
|
+
Whether a UTF-16 code unit is a space, a tab, a line feed, or the end.
|
|
16
|
+
*/
|
|
17
|
+
function isSpaceOrLineEnd(code) {
|
|
18
|
+
return isSpace(code) || code === LF || Number.isNaN(code);
|
|
19
|
+
}
|
|
20
|
+
/*
|
|
21
|
+
The lookup tables are maps rather than objects, so that a property added to `Object.prototype` is never read as an entry.
|
|
22
|
+
*/
|
|
23
|
+
const RADIX_DIGIT = new Map([
|
|
24
|
+
['x', code => (code >= 0x30 && code <= 0x39) || (code >= 0x41 && code <= 0x46)],
|
|
25
|
+
['o', code => code >= 0x30 && code <= 0x37],
|
|
26
|
+
['b', code => code === 0x30 || code === 0x31],
|
|
27
|
+
]);
|
|
28
|
+
/*
|
|
29
|
+
The longest token that the instant regular expression and the diagnostic regular expressions look at. Their repeated groups need stack in proportion to the input, so a longer token is rejected with a general message, or described from its first part.
|
|
33
30
|
*/
|
|
34
31
|
const MAX_DIAGNOSED_LENGTH = 1000;
|
|
35
32
|
const RADIX_NAME = { x: 'hexadecimal', o: 'octal', b: 'binary' };
|
|
36
33
|
const RADIX_ARTICLE = { x: 'A', o: 'An', b: 'A' };
|
|
37
34
|
const INSTANT_PREFIX = /^\d{4}-\d{2}-\d{2}/v;
|
|
38
|
-
const INSTANT_START = /^\d{4}-\d{2}-\d{2}T\d{2}:\d/v;
|
|
39
35
|
const DURATION_UNIT_NAMES = DURATION_UNITS.keys().toArray();
|
|
40
36
|
const INSTANT = /^(?<year>\d{4})-(?<month>\d{2})-(?<day>\d{2})T(?<hour>\d{2}):(?<minute>\d{2}):(?<second>\d{2})(?:\.(?<fraction>\d+))?(?<offset>Z|[+\-](?<offsetHour>\d{2}):(?<offsetMinute>\d{2}))$/v;
|
|
37
|
+
const LOCAL_DATE_TIME = /^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d+)?$/v;
|
|
38
|
+
const INSTANT_FORMAT = 'An instant is written as 2026-09-19T14:00:00Z, with an optional fraction of up to nine digits and an offset of Z or ±HH:MM';
|
|
41
39
|
const UNICODE_ESCAPE = /u\{(?<hex>[\da-f]{1,6})\}/vy;
|
|
42
40
|
/*
|
|
43
41
|
Every C0 control character except tab and line feed, and DEL. They are errors anywhere in a document.
|
|
@@ -45,14 +43,23 @@ Every C0 control character except tab and line feed, and DEL. They are errors an
|
|
|
45
43
|
// eslint-disable-next-line no-control-regex, regexp/no-control-character -- Control characters are exactly what must be found.
|
|
46
44
|
const CONTROL_CHARACTER = /[\u{0}-\u{8}\u{B}-\u{1F}\u{7F}]/v;
|
|
47
45
|
const LITERAL_STRING_END = /[\n']/gv;
|
|
48
|
-
const
|
|
46
|
+
const KEY_QUOTING_HINT = '. A key that contains characters other than letters, digits, “_”, and “-” must be quoted';
|
|
47
|
+
/*
|
|
48
|
+
The characters that may be meant as part of a key: visible ones that have no meaning in the grammar, which includes every bare key character. A `=` is left out, because after a key, as in `name=foo`, it was most likely meant as the `:` of INI and TOML, so a key that holds a `=` gets no quoting hint.
|
|
49
|
+
*/
|
|
50
|
+
const QUOTABLE_KEY_CHARACTER = String.raw `[^\p{Default_Ignorable_Code_Point}\p{Other}\p{White_Space}"#'*,.\/:=\[\]\{\}]`;
|
|
51
|
+
const QUOTABLE_KEY_START = new RegExp(`^${QUOTABLE_KEY_CHARACTER}`, 'v');
|
|
52
|
+
/*
|
|
53
|
+
A key made of them, up to its `:`, so the quoting hint can show it quoted.
|
|
54
|
+
*/
|
|
55
|
+
const QUOTABLE_KEY = new RegExp(`^${QUOTABLE_KEY_CHARACTER}+(?=:)`, 'v');
|
|
49
56
|
const ESCAPED_STRING_SPECIAL = /[\n"\\]/gv;
|
|
50
|
-
const SIMPLE_ESCAPES =
|
|
51
|
-
'\\'
|
|
52
|
-
'"'
|
|
53
|
-
n
|
|
54
|
-
t
|
|
55
|
-
|
|
57
|
+
const SIMPLE_ESCAPES = new Map([
|
|
58
|
+
['\\', '\\'],
|
|
59
|
+
['"', '"'],
|
|
60
|
+
['n', '\n'],
|
|
61
|
+
['t', '\t'],
|
|
62
|
+
]);
|
|
56
63
|
/*
|
|
57
64
|
What a backslash followed by one of these characters was probably meant to be.
|
|
58
65
|
*/
|
|
@@ -61,13 +68,34 @@ const ESCAPE_MISTAKES = new Map([
|
|
|
61
68
|
['\'', 'A \' needs no escape inside "..."'],
|
|
62
69
|
['\n', String.raw `A backslash must be followed by an escape character. Use \\ for a literal backslash, or a '...' string`],
|
|
63
70
|
]);
|
|
64
|
-
export function parse(text, options
|
|
65
|
-
const
|
|
66
|
-
validateIntegersOption(integers);
|
|
71
|
+
export function parse(text, options) {
|
|
72
|
+
const integers = getIntegersOption(options);
|
|
67
73
|
const source = decode(text);
|
|
68
74
|
checkCharacters(source);
|
|
75
|
+
// The default `createTime` makes a `Temporal` object of every instant and duration, so no `Time` is left.
|
|
69
76
|
return new Parser(source, integers).parseDocument();
|
|
70
77
|
}
|
|
78
|
+
/*
|
|
79
|
+
An instant, as nanoseconds since the Unix epoch, or a duration, as its length in nanoseconds. A class, so that a dotted key sees it as a value and not as an object.
|
|
80
|
+
*/
|
|
81
|
+
export class Time {
|
|
82
|
+
type;
|
|
83
|
+
nanoseconds;
|
|
84
|
+
constructor(type, nanoseconds) {
|
|
85
|
+
this.type = type;
|
|
86
|
+
this.nanoseconds = nanoseconds;
|
|
87
|
+
}
|
|
88
|
+
toTemporal() {
|
|
89
|
+
return this.type === 'Instant' ? new (requireTemporal().Instant)(this.nanoseconds) : createDuration(this.nanoseconds);
|
|
90
|
+
}
|
|
91
|
+
}
|
|
92
|
+
/*
|
|
93
|
+
The same as `parse()` for a string, except that each instant and duration is a `Time` instead of a `Temporal` object, so that it works without `Temporal`. For the tree, which makes the `Temporal` object only when the value is read.
|
|
94
|
+
*/
|
|
95
|
+
export function parseWithTimes(text) {
|
|
96
|
+
checkCharacters(text);
|
|
97
|
+
return new Parser(text, 'bigint', time => time).parseDocument();
|
|
98
|
+
}
|
|
71
99
|
const typedArrayTag = Object.getOwnPropertyDescriptor(Object.getPrototypeOf(Uint8Array.prototype), Symbol.toStringTag).get;
|
|
72
100
|
function decode(text) {
|
|
73
101
|
if (typeof text === 'string') {
|
|
@@ -178,19 +206,25 @@ function describeValue(value) {
|
|
|
178
206
|
return value !== null && typeof value === 'object' && Object.getPrototypeOf(value) === Object.prototype ? 'an object' : 'a value';
|
|
179
207
|
}
|
|
180
208
|
function formatPath(path) {
|
|
181
|
-
return path.map(segment =>
|
|
209
|
+
return path.map(segment => describeKey(segment)).join('.');
|
|
182
210
|
}
|
|
183
211
|
class Parser {
|
|
184
212
|
#source;
|
|
185
213
|
#index = 0;
|
|
186
214
|
#integers;
|
|
215
|
+
#createTime;
|
|
187
216
|
/*
|
|
188
217
|
Objects created by dotted keys. They may be extended by further dotted keys, while an object written with braces is closed.
|
|
189
218
|
*/
|
|
190
219
|
#dottedObjects = new WeakSet();
|
|
191
|
-
|
|
220
|
+
/*
|
|
221
|
+
Where the document's collection starts, after the comments and whitespace before it.
|
|
222
|
+
*/
|
|
223
|
+
#documentStart = 0;
|
|
224
|
+
constructor(source, integers, createTime = time => time.toTemporal()) {
|
|
192
225
|
this.#source = source;
|
|
193
226
|
this.#integers = integers;
|
|
227
|
+
this.#createTime = createTime;
|
|
194
228
|
}
|
|
195
229
|
#fail(reason, offset = this.#index) {
|
|
196
230
|
throw ParseError.create(reason, this.#source, offset);
|
|
@@ -218,11 +252,11 @@ class Parser {
|
|
|
218
252
|
}
|
|
219
253
|
let hasLineBreak = this.#skipTrivia();
|
|
220
254
|
while (!this.#isAtEnd()) {
|
|
255
|
+
if (this.#code() === COMMA) {
|
|
256
|
+
this.#fail('Top-level entries are separated by line breaks, not commas. Use braces for a one-line object');
|
|
257
|
+
}
|
|
221
258
|
if (!hasLineBreak) {
|
|
222
|
-
|
|
223
|
-
this.#fail('Top-level entries are separated by line breaks, not commas. Use braces for a one-line object');
|
|
224
|
-
}
|
|
225
|
-
this.#fail(`Expected a line break before the next entry, but found ${this.#describeHere()}`);
|
|
259
|
+
this.#fail(`Expected a line break before the next entry, but found ${this.#describeHere()}${this.#slashCommentHint()}`);
|
|
226
260
|
}
|
|
227
261
|
this.#parseEntry(object, 1);
|
|
228
262
|
hasLineBreak = this.#skipTrivia();
|
|
@@ -230,21 +264,18 @@ class Parser {
|
|
|
230
264
|
return object;
|
|
231
265
|
}
|
|
232
266
|
/*
|
|
233
|
-
A
|
|
267
|
+
A whole document that is one scalar, such as `5` or an instant, fails as an entry. That failure is reported as what it is.
|
|
234
268
|
*/
|
|
235
269
|
#diagnoseBareValue(start) {
|
|
236
270
|
this.#index = start;
|
|
237
271
|
let isBareValue = false;
|
|
238
272
|
try {
|
|
239
273
|
this.#parseValue(1);
|
|
240
|
-
|
|
274
|
+
this.#skipTrivia();
|
|
275
|
+
isBareValue = this.#isAtEnd();
|
|
241
276
|
}
|
|
242
|
-
catch
|
|
243
|
-
//
|
|
244
|
-
if (INSTANT_START.test(this.#source.slice(start, start + 15))) {
|
|
245
|
-
throw error;
|
|
246
|
-
}
|
|
247
|
-
// Otherwise it is not a bare value either, and the original error stands.
|
|
277
|
+
catch {
|
|
278
|
+
// Not a bare value either, so the original error stands.
|
|
248
279
|
}
|
|
249
280
|
if (isBareValue) {
|
|
250
281
|
this.#fail('A bare value is not a document. A document is an object or an array, so write it as `key: value` or `[value]`', start);
|
|
@@ -258,7 +289,7 @@ class Parser {
|
|
|
258
289
|
let hasCrossedLineBreak = false;
|
|
259
290
|
for (;;) {
|
|
260
291
|
const code = source.charCodeAt(this.#index);
|
|
261
|
-
if (code
|
|
292
|
+
if (isSpace(code)) {
|
|
262
293
|
this.#index++;
|
|
263
294
|
}
|
|
264
295
|
else if (code === LF) {
|
|
@@ -296,7 +327,7 @@ class Parser {
|
|
|
296
327
|
const nested = source.indexOf('/*', start + 2);
|
|
297
328
|
// The body may not contain `/*`. An opening that overlaps the closing `*/`, as in `/*/`, is not inside the body.
|
|
298
329
|
if (nested !== -1 && nested + 2 <= end) {
|
|
299
|
-
this.#fail('Block comments cannot be nested, and their body may not contain
|
|
330
|
+
this.#fail('Block comments cannot be nested, and their body may not contain “/*”', nested);
|
|
300
331
|
}
|
|
301
332
|
this.#index = end + 2;
|
|
302
333
|
}
|
|
@@ -307,47 +338,152 @@ class Parser {
|
|
|
307
338
|
const keyStart = this.#index;
|
|
308
339
|
const path = this.#parseKey();
|
|
309
340
|
if (depth + path.length - 1 > MAX_DEPTH) {
|
|
310
|
-
this.#
|
|
341
|
+
this.#failTooDeep(keyStart);
|
|
311
342
|
}
|
|
312
343
|
const code = this.#code();
|
|
313
344
|
if (code !== COLON) {
|
|
314
345
|
this.#failMissingColon(keyStart);
|
|
315
346
|
}
|
|
347
|
+
const colon = this.#index;
|
|
316
348
|
this.#index++;
|
|
317
349
|
this.#skipTrivia();
|
|
318
|
-
|
|
350
|
+
let value;
|
|
351
|
+
try {
|
|
352
|
+
value = this.#parseValue(depth + path.length);
|
|
353
|
+
}
|
|
354
|
+
catch (error) {
|
|
355
|
+
if (error instanceof ParseError) {
|
|
356
|
+
this.#diagnoseBadValue(error, path, keyStart, colon);
|
|
357
|
+
}
|
|
358
|
+
throw error;
|
|
359
|
+
}
|
|
319
360
|
this.#assign(object, path, value, keyStart);
|
|
320
361
|
}
|
|
321
|
-
|
|
362
|
+
/*
|
|
363
|
+
Reports a value that failed to parse as what the entry was meant to be, when that is clear.
|
|
364
|
+
*/
|
|
365
|
+
#diagnoseBadValue(error, path, keyStart, colon) {
|
|
366
|
+
// A key that contains a `:`, as in `12:30: 'lunch'`, ends at the first `:`, and the rest is read as the value.
|
|
367
|
+
if (isBareKeyCharacter(this.#code(colon + 1))) {
|
|
368
|
+
this.#diagnoseKeyWithColon(keyStart);
|
|
369
|
+
}
|
|
370
|
+
this.#index = colon + 1;
|
|
371
|
+
const isValueOnNextLine = this.#skipTrivia();
|
|
372
|
+
const valueStart = this.#index;
|
|
373
|
+
const commentHint = this.#describeCommentAsValue(colon);
|
|
374
|
+
if (isValueOnNextLine) {
|
|
375
|
+
this.#diagnoseMissingValue(valueStart, path, keyStart, commentHint);
|
|
376
|
+
}
|
|
377
|
+
// A value that was left out at the end of the document or of an object.
|
|
378
|
+
const code = this.#code(valueStart);
|
|
379
|
+
if (commentHint !== '' && error.offset === valueStart && (code === CLOSE_BRACE || Number.isNaN(code))) {
|
|
380
|
+
this.#fail(`${error.reason}${commentHint}`, valueStart);
|
|
381
|
+
}
|
|
382
|
+
}
|
|
383
|
+
/*
|
|
384
|
+
A bare key followed directly by `:` and more of a key, up to a `:` that ends the key, as in `a:b: 1`.
|
|
385
|
+
*/
|
|
386
|
+
#diagnoseKeyWithColon(keyStart) {
|
|
322
387
|
const source = this.#source;
|
|
323
|
-
let
|
|
324
|
-
|
|
325
|
-
|
|
388
|
+
for (let index = keyStart; index < keyStart + MAX_DIAGNOSED_LENGTH; index++) {
|
|
389
|
+
const code = source.charCodeAt(index);
|
|
390
|
+
if (code === COLON) {
|
|
391
|
+
const next = source.charCodeAt(index + 1);
|
|
392
|
+
if (next === LF || isSpace(next) || Number.isNaN(next)) {
|
|
393
|
+
this.#fail(`A key that contains “:” must be quoted, as in '${abbreviate(source.slice(keyStart, index))}'`, keyStart);
|
|
394
|
+
}
|
|
395
|
+
}
|
|
396
|
+
else if (!isBareKeyCharacter(code)) {
|
|
397
|
+
return;
|
|
398
|
+
}
|
|
326
399
|
}
|
|
400
|
+
}
|
|
401
|
+
/*
|
|
402
|
+
The hint for a `#` directly after a `:`, as in `color: #FFF`, which starts a comment rather than a value.
|
|
403
|
+
*/
|
|
404
|
+
#describeCommentAsValue(colon) {
|
|
405
|
+
const source = this.#source;
|
|
406
|
+
const hash = skipSpaces(source, colon + 1);
|
|
407
|
+
if (source.charCodeAt(hash) !== HASH) {
|
|
408
|
+
return '';
|
|
409
|
+
}
|
|
410
|
+
// A `#` followed by a space, or by another `#`, starts an ordinary comment. A separator or a closing bracket after the value is not part of it. The regular expression, which needs stack in proportion to its match, only sees the start of a long comment, which is cut short in the message anyway.
|
|
411
|
+
const match = /^#[^\t\n #,\]\}][^\t\n ,\]\}]*/v.exec(source.slice(hash, hash + MAX_DIAGNOSED_LENGTH));
|
|
412
|
+
return match === null ? '' : `. “#” starts a comment, so a value that starts with “#” must be quoted${quotingExample(match[0])}`;
|
|
413
|
+
}
|
|
414
|
+
/*
|
|
415
|
+
An entry whose value was left out, as in `a:` followed by `'b': 1` on the next line, reads the next key as the value. That failure is reported as what it is.
|
|
416
|
+
*/
|
|
417
|
+
#diagnoseMissingValue(start, path, keyStart, commentHint) {
|
|
418
|
+
// A YAML block sequence.
|
|
419
|
+
if (this.#code(start) === DASH && isSpace(this.#code(start + 1))) {
|
|
420
|
+
this.#fail('Expected a value, but found a “-” list. An array is written in brackets, as in [80, 443]', start);
|
|
421
|
+
}
|
|
422
|
+
this.#index = start;
|
|
423
|
+
let key;
|
|
424
|
+
try {
|
|
425
|
+
key = this.#parseKey();
|
|
426
|
+
}
|
|
427
|
+
catch {
|
|
428
|
+
// Not a key either, so the original error stands.
|
|
429
|
+
}
|
|
430
|
+
if (key === undefined || this.#code() !== COLON) {
|
|
431
|
+
return;
|
|
432
|
+
}
|
|
433
|
+
// Whitespace or the end follows the `:` of a bare key. A digit follows the `:` in an instant such as `2026-09-19T25:00:00Z`, whose own error is more precise. A quoted key cannot be part of a value, so anything may follow its `:`.
|
|
434
|
+
const next = this.#code(this.#index + 1);
|
|
435
|
+
const firstCode = this.#code(start);
|
|
436
|
+
if (firstCode !== SINGLE_QUOTE && firstCode !== DOUBLE_QUOTE && next !== LF && !isSpace(next) && !Number.isNaN(next)) {
|
|
437
|
+
return;
|
|
438
|
+
}
|
|
439
|
+
const hint = commentHint === '' ? this.#describeIndentedKey(path, key, keyStart, start) : commentHint;
|
|
440
|
+
this.#fail(`Expected a value, but found the key ${abbreviate(formatPath(key), 200)}${hint}`, start);
|
|
441
|
+
}
|
|
442
|
+
/*
|
|
443
|
+
The hint for a key at `start` that is indented under the entry whose value is missing, as YAML nests an object.
|
|
444
|
+
*/
|
|
445
|
+
#describeIndentedKey(path, key, keyStart, start) {
|
|
446
|
+
const keyIndentation = getIndentation(this.#source, keyStart);
|
|
447
|
+
const indentation = getIndentation(this.#source, start);
|
|
448
|
+
if (keyIndentation === undefined || indentation === undefined || indentation <= keyIndentation) {
|
|
449
|
+
return '';
|
|
450
|
+
}
|
|
451
|
+
// The keys are written as in a document, so that the suggestion is valid, rather than as JSON strings like the rest of the message, whose escapes, such as `\b`, are not all valid.
|
|
452
|
+
const parent = abbreviate(path.map(segment => formatKey(segment)).join('.'), 200);
|
|
453
|
+
const child = abbreviate(key.map(segment => formatKey(segment)).join('.'), 200);
|
|
454
|
+
return `. Indentation does not nest objects, so write ${parent}: {${child}: …} or ${parent}.${child}: …`;
|
|
455
|
+
}
|
|
456
|
+
#failMissingColon(keyStart) {
|
|
457
|
+
const source = this.#source;
|
|
458
|
+
const next = skipSpaces(source, this.#index);
|
|
327
459
|
const nextCode = source.charCodeAt(next);
|
|
328
460
|
if (nextCode === COLON && next > this.#index) {
|
|
329
|
-
this.#fail('Whitespace is not allowed between a key and its
|
|
461
|
+
this.#fail('Whitespace is not allowed between a key and its “:”');
|
|
330
462
|
}
|
|
331
|
-
//
|
|
463
|
+
// A block comment followed by the ":" was meant to come before it.
|
|
332
464
|
if (nextCode === SLASH && source.charCodeAt(next + 1) === ASTERISK) {
|
|
333
|
-
|
|
465
|
+
const commentEnd = source.indexOf('*/', next + 2);
|
|
466
|
+
if (commentEnd !== -1 && source.charCodeAt(skipSpaces(source, commentEnd + 2)) === COLON) {
|
|
467
|
+
this.#fail('A comment is not allowed between a key and its “:”', next);
|
|
468
|
+
}
|
|
334
469
|
}
|
|
335
470
|
const colon = source.indexOf(':', next);
|
|
336
|
-
// A key with a space in it, such as `the name: 1`. The hint is only given for plain words, because quoting a dotted or quoted key would change what it means.
|
|
337
|
-
if (colon !== -1 && next > this.#index && isBareKeyCharacter(nextCode) && colon < findLineEnd(source, next)) {
|
|
338
|
-
|
|
471
|
+
// A key with a space in it, such as `the name: 1`. The hint is only given for plain words, because quoting a dotted or quoted key would change what it means, and for a `:` that whitespace or the end follows, because quoting the words before the `:` in `server localhost:8080` would give a valid document with another meaning. So `the name:1` gets no hint.
|
|
472
|
+
if (colon !== -1 && next > this.#index && isBareKeyCharacter(nextCode) && colon < findLineEnd(source, next) && isSpaceOrLineEnd(source.charCodeAt(colon + 1))) {
|
|
473
|
+
// Only spaces and tabs are trimmed, because `trimEnd()` would also remove characters that are not whitespace in SOML, such as U+00A0.
|
|
474
|
+
const key = source.slice(keyStart, skipSpacesBack(source, colon));
|
|
339
475
|
if (isWordsWithSpaces(key)) {
|
|
340
476
|
this.#fail(`A bare key cannot contain spaces. Quote it, as in '${abbreviate(key)}'`, keyStart);
|
|
341
477
|
}
|
|
342
478
|
}
|
|
343
479
|
if (this.#isAtEnd() || this.#code() === LF) {
|
|
344
|
-
this.#fail('Expected
|
|
480
|
+
this.#fail('Expected “:” after the key');
|
|
345
481
|
}
|
|
346
482
|
// A character directly after a bare key is most likely meant to be part of it.
|
|
347
483
|
const previousCode = source.charCodeAt(this.#index - 1);
|
|
348
|
-
const hint = previousCode !== SINGLE_QUOTE && previousCode !== DOUBLE_QUOTE && next === this.#index
|
|
484
|
+
const hint = previousCode !== SINGLE_QUOTE && previousCode !== DOUBLE_QUOTE && next === this.#index && QUOTABLE_KEY_START.test(source.slice(next, next + 2)) ? KEY_QUOTING_HINT : '';
|
|
349
485
|
this.#index = next;
|
|
350
|
-
this.#fail(`Expected
|
|
486
|
+
this.#fail(`Expected “:” after the key, but found ${this.#describeHere()}${hint}`);
|
|
351
487
|
}
|
|
352
488
|
#parseKey() {
|
|
353
489
|
const path = [this.#parseKeySegment()];
|
|
@@ -373,14 +509,26 @@ class Parser {
|
|
|
373
509
|
end++;
|
|
374
510
|
}
|
|
375
511
|
if (end === start) {
|
|
376
|
-
if (
|
|
377
|
-
this.#
|
|
512
|
+
if (code === OPEN_BRACKET) {
|
|
513
|
+
this.#diagnoseTableHeader();
|
|
378
514
|
}
|
|
379
|
-
|
|
515
|
+
const expected = isAfterDot ? 'Expected a key segment after “.”' : 'Expected a key';
|
|
516
|
+
this.#fail(this.#isAtEnd() ? expected : `${expected}, but found ${this.#describeHere()}${this.#keyQuotingHint()}${this.#slashCommentHint()}`);
|
|
380
517
|
}
|
|
381
518
|
this.#index = end;
|
|
382
519
|
return source.slice(start, end);
|
|
383
520
|
}
|
|
521
|
+
/*
|
|
522
|
+
The hint for a key that starts with a character a bare key cannot hold, such as the `$` in `$schema`, or nothing for a character with another meaning, such as `}`.
|
|
523
|
+
*/
|
|
524
|
+
#keyQuotingHint() {
|
|
525
|
+
if (!QUOTABLE_KEY_START.test(this.#source.slice(this.#index, this.#index + 2))) {
|
|
526
|
+
return '';
|
|
527
|
+
}
|
|
528
|
+
// A longer key gets the hint without the example, and the regular expression, which backtracks, never sees a huge run.
|
|
529
|
+
const key = QUOTABLE_KEY.exec(this.#source.slice(this.#index, this.#index + MAX_DIAGNOSED_LENGTH))?.[0];
|
|
530
|
+
return key === undefined ? KEY_QUOTING_HINT : `${KEY_QUOTING_HINT}, as in '${abbreviate(key)}'`;
|
|
531
|
+
}
|
|
384
532
|
#assign(object, path, value, keyStart) {
|
|
385
533
|
let target = object;
|
|
386
534
|
const lastIndex = path.length - 1;
|
|
@@ -412,61 +560,59 @@ class Parser {
|
|
|
412
560
|
}
|
|
413
561
|
defineMember(target, key, value);
|
|
414
562
|
}
|
|
563
|
+
#failTooDeep(offset = this.#index) {
|
|
564
|
+
this.#fail(`The document is nested more than ${MAX_DEPTH} levels deep`, offset);
|
|
565
|
+
}
|
|
415
566
|
#parseObject(depth) {
|
|
416
|
-
if (depth > MAX_DEPTH) {
|
|
417
|
-
this.#fail(`The document is nested more than ${MAX_DEPTH} levels deep`);
|
|
418
|
-
}
|
|
419
|
-
const start = this.#index;
|
|
420
567
|
const object = {};
|
|
421
|
-
this.#
|
|
422
|
-
this.#skipTrivia();
|
|
423
|
-
for (;;) {
|
|
424
|
-
const code = this.#code();
|
|
425
|
-
if (code === CLOSE_BRACE) {
|
|
426
|
-
this.#index++;
|
|
427
|
-
return object;
|
|
428
|
-
}
|
|
429
|
-
if (this.#isAtEnd()) {
|
|
430
|
-
this.#fail('Unterminated object: expected "}"', start);
|
|
431
|
-
}
|
|
568
|
+
this.#parseItems(depth, CLOSE_BRACE, () => {
|
|
432
569
|
this.#parseEntry(object, depth);
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
this.#index++;
|
|
436
|
-
this.#skipTrivia();
|
|
437
|
-
}
|
|
438
|
-
else if (this.#code() !== CLOSE_BRACE) {
|
|
439
|
-
this.#fail(this.#isAtEnd() ? 'Unterminated object: expected "}"' : `Expected "," or "}" after an object member, but found ${this.#describeHere()}`, this.#isAtEnd() ? start : this.#index);
|
|
440
|
-
}
|
|
441
|
-
}
|
|
570
|
+
});
|
|
571
|
+
return object;
|
|
442
572
|
}
|
|
443
573
|
#parseArray(depth) {
|
|
574
|
+
const array = [];
|
|
575
|
+
this.#parseItems(depth, CLOSE_BRACKET, () => {
|
|
576
|
+
array.push(this.#parseValue(depth + 1));
|
|
577
|
+
});
|
|
578
|
+
return array;
|
|
579
|
+
}
|
|
580
|
+
/*
|
|
581
|
+
`{` or `[` at depth `depth`, then items separated by a comma, a line break, or both, with an optional trailing comma, then the `closing` bracket. A comma must be on the line of the item before it.
|
|
582
|
+
*/
|
|
583
|
+
#parseItems(depth, closing, parseItem) {
|
|
444
584
|
if (depth > MAX_DEPTH) {
|
|
445
|
-
this.#
|
|
585
|
+
this.#failTooDeep();
|
|
446
586
|
}
|
|
447
587
|
const start = this.#index;
|
|
448
|
-
const array = [];
|
|
449
588
|
this.#index++;
|
|
450
589
|
this.#skipTrivia();
|
|
451
|
-
|
|
452
|
-
|
|
453
|
-
|
|
454
|
-
|
|
455
|
-
|
|
590
|
+
while (this.#code() !== closing) {
|
|
591
|
+
let hasLineBreak = false;
|
|
592
|
+
let itemEnd = this.#index;
|
|
593
|
+
if (!this.#isAtEnd()) {
|
|
594
|
+
parseItem();
|
|
595
|
+
itemEnd = this.#index;
|
|
596
|
+
hasLineBreak = this.#skipTrivia();
|
|
456
597
|
}
|
|
457
598
|
if (this.#isAtEnd()) {
|
|
458
|
-
this.#fail(
|
|
599
|
+
this.#fail(`Unterminated ${closing === CLOSE_BRACE ? 'object' : 'array'}: expected “${String.fromCharCode(closing)}”`, start);
|
|
459
600
|
}
|
|
460
|
-
array.push(this.#parseValue(depth + 1));
|
|
461
|
-
this.#skipTrivia();
|
|
462
601
|
if (this.#code() === COMMA) {
|
|
602
|
+
if (hasLineBreak) {
|
|
603
|
+
this.#fail('A comma must be on the same line as the item before it. The line break already separates the items, so remove the comma');
|
|
604
|
+
}
|
|
463
605
|
this.#index++;
|
|
464
606
|
this.#skipTrivia();
|
|
465
607
|
}
|
|
466
|
-
else if (this.#code() !==
|
|
467
|
-
|
|
608
|
+
else if (!hasLineBreak && this.#code() !== closing) {
|
|
609
|
+
const item = closing === CLOSE_BRACE ? 'an object member' : 'an array item';
|
|
610
|
+
// Without a line break that separates, a line break in the gap is inside a block comment.
|
|
611
|
+
const hint = this.#source.slice(itemEnd, this.#index).includes('\n') ? '. A line break inside a block comment does not separate items' : this.#slashCommentHint();
|
|
612
|
+
this.#fail(`Expected “,”, a line break, or “${String.fromCharCode(closing)}” after ${item}, but found ${this.#describeHere()}${hint}`);
|
|
468
613
|
}
|
|
469
614
|
}
|
|
615
|
+
this.#index++;
|
|
470
616
|
}
|
|
471
617
|
#parseValue(depth) {
|
|
472
618
|
const code = this.#code();
|
|
@@ -476,7 +622,10 @@ class Parser {
|
|
|
476
622
|
if (code === OPEN_BRACKET) {
|
|
477
623
|
return this.#parseArray(depth);
|
|
478
624
|
}
|
|
625
|
+
const start = this.#index;
|
|
479
626
|
let value;
|
|
627
|
+
// An int or a float written with digits.
|
|
628
|
+
let isNumber = false;
|
|
480
629
|
if (code === SINGLE_QUOTE || code === DOUBLE_QUOTE) {
|
|
481
630
|
value = this.#parseString();
|
|
482
631
|
}
|
|
@@ -497,16 +646,55 @@ class Parser {
|
|
|
497
646
|
}
|
|
498
647
|
else if (code === DASH || isDigit(code)) {
|
|
499
648
|
value = this.#parseNumberOrInstant();
|
|
649
|
+
isNumber = typeof value === 'bigint' || typeof value === 'number';
|
|
500
650
|
}
|
|
501
651
|
else {
|
|
502
652
|
this.#failUnexpectedValue();
|
|
503
653
|
}
|
|
504
|
-
|
|
505
|
-
|
|
506
|
-
|
|
654
|
+
if (!isValueEnd(this.#code())) {
|
|
655
|
+
this.#failAfterValue(start, isNumber);
|
|
656
|
+
}
|
|
657
|
+
if (isNumber && isSpace(this.#code())) {
|
|
658
|
+
this.#diagnoseUnitAfterSpace(start);
|
|
507
659
|
}
|
|
508
660
|
return value;
|
|
509
661
|
}
|
|
662
|
+
/*
|
|
663
|
+
A character that cannot follow the value that starts at `start`.
|
|
664
|
+
*/
|
|
665
|
+
#failAfterValue(start, isNumber) {
|
|
666
|
+
const code = this.#code();
|
|
667
|
+
// Go writes microseconds as `µs`, with the micro sign or the Greek letter mu. The duration is only suggested when it is valid, which a number with an exponent or a radix prefix, for example, is not.
|
|
668
|
+
if (isNumber && (code === 0xB5 || code === 0x3_BC) && this.#code(this.#index + 1) === 0x73 /* s */) {
|
|
669
|
+
const number = this.#source.slice(start, this.#index);
|
|
670
|
+
this.#fail(`The unit for microseconds is written us${isValidValue(`${number}us`) ? `, as in ${abbreviate(number)}us` : ''}`);
|
|
671
|
+
}
|
|
672
|
+
// A `''` inside a '...' string, as SQL and YAML escape a quote, ends the string.
|
|
673
|
+
if (code === SINGLE_QUOTE && this.#code(start) === SINGLE_QUOTE) {
|
|
674
|
+
this.#fail('There is no \'\' escape in a \'...\' string. Write a string that contains \' as "...", as in "it\'s"');
|
|
675
|
+
}
|
|
676
|
+
this.#fail(`Unexpected ${this.#describeHere()} after a value`);
|
|
677
|
+
}
|
|
678
|
+
/*
|
|
679
|
+
A number followed by a space and a word, such as `512 MiB` or `10 seconds`, which is an error anyway. A word followed by more than spaces, a separator, or a comment is left to the general errors, because it may be a key, as in `a: 1 b: 2`, or a sentence.
|
|
680
|
+
*/
|
|
681
|
+
#diagnoseUnitAfterSpace(start) {
|
|
682
|
+
const source = this.#source;
|
|
683
|
+
const unitStart = skipSpaces(source, this.#index);
|
|
684
|
+
let unitEnd = unitStart;
|
|
685
|
+
while (isLetter(source.charCodeAt(unitEnd))) {
|
|
686
|
+
unitEnd++;
|
|
687
|
+
}
|
|
688
|
+
const unit = source.slice(unitStart, unitEnd);
|
|
689
|
+
// A keyword after a number is a missing comma.
|
|
690
|
+
if (unit === '' || !isValueEnd(source.charCodeAt(skipSpaces(source, unitEnd))) || ['true', 'false', 'null', 'infinity'].includes(unit)) {
|
|
691
|
+
return;
|
|
692
|
+
}
|
|
693
|
+
const number = source.slice(start, this.#index);
|
|
694
|
+
// The duration is only suggested when it is valid, which a number with an exponent or a radix prefix, a fraction of a nanosecond, a negative zero, or a value outside the 64-bit range is not. The string is always offered, because a unit such as `m` may mean meters rather than minutes.
|
|
695
|
+
const duration = DURATION_UNITS.has(unit) && isValidValue(`${number}${unit}`) ? `${abbreviate(number)}${unit}` : '10s';
|
|
696
|
+
this.#fail(`A unit cannot follow a number after a space. Write a duration without the space, as in ${duration}, and anything else, such as a size, as a string, as in '${abbreviate(`${number} ${unit}`)}'`, unitStart);
|
|
697
|
+
}
|
|
510
698
|
#isKeyword(word) {
|
|
511
699
|
if (!this.#source.startsWith(word, this.#index)) {
|
|
512
700
|
return false;
|
|
@@ -525,10 +713,11 @@ class Parser {
|
|
|
525
713
|
}
|
|
526
714
|
const code = this.#code();
|
|
527
715
|
if (code === 0x2B /* + */) {
|
|
528
|
-
this.#fail('A
|
|
716
|
+
this.#fail('A “+” sign is not allowed. A number without a sign is positive');
|
|
529
717
|
}
|
|
718
|
+
// A `.` that no digit follows begins a string, such as `.env` or `./foo`, rather than a number.
|
|
530
719
|
if (code === DOT) {
|
|
531
|
-
this.#fail('A number cannot begin with
|
|
720
|
+
this.#fail(isDigit(this.#code(this.#index + 1)) ? 'A number cannot begin with “.”; write a digit before it, as in 0.5' : describeUnknownWord('.', this.#unquotedText()));
|
|
532
721
|
}
|
|
533
722
|
let wordEnd = this.#index;
|
|
534
723
|
while (isBareKeyCharacter(this.#code(wordEnd))) {
|
|
@@ -536,13 +725,66 @@ class Parser {
|
|
|
536
725
|
}
|
|
537
726
|
if (wordEnd > this.#index) {
|
|
538
727
|
const word = this.#source.slice(this.#index, wordEnd);
|
|
539
|
-
//
|
|
540
|
-
|
|
728
|
+
// A key where a value should be, as when the value of an entry is left out and the next line has a key, or as in `[a: 1]`, is reported as a key. A word after the `:` of a member and a space, as in `msg: Error: file not found` or `url: https://example.com`, is that member's value instead, an unquoted string. Without the space, as in `a:b: 1`, the first `:` was most likely meant as part of the key.
|
|
729
|
+
const spacesStart = skipSpacesBack(this.#source, this.#index);
|
|
730
|
+
const isMemberValue = spacesStart < this.#index && this.#code(spacesStart - 1) === COLON;
|
|
731
|
+
if (!isMemberValue && this.#code(wordEnd) === COLON) {
|
|
541
732
|
this.#fail(`Expected a value, but found the key ${abbreviate(word)}`);
|
|
542
733
|
}
|
|
543
|
-
this.#
|
|
734
|
+
this.#diagnoseTableHeader(true);
|
|
735
|
+
this.#fail(describeUnknownWord(word, this.#unquotedText()));
|
|
736
|
+
}
|
|
737
|
+
this.#fail(`Expected a value, but found ${this.#describeHere()}${this.#isBlockScalarIndicator() ? '. Write a multiline string as a block string, between \'\'\' lines' : this.#slashCommentHint()}`);
|
|
738
|
+
}
|
|
739
|
+
/*
|
|
740
|
+
The text from `start` that was most likely meant as one unquoted string, such as `John Smith`: up to the end of the line, a comma, a closing bracket, or a comment after a space or a tab, without the spaces and tabs at its end.
|
|
741
|
+
*/
|
|
742
|
+
#unquotedText(start = this.#index) {
|
|
743
|
+
const source = this.#source;
|
|
744
|
+
const limit = Math.min(source.length, start + MAX_DIAGNOSED_LENGTH);
|
|
745
|
+
let end = start;
|
|
746
|
+
while (end < limit && !isUnquotedTextEnd(source, end)) {
|
|
747
|
+
end++;
|
|
748
|
+
}
|
|
749
|
+
return source.slice(start, skipSpacesBack(source, end));
|
|
750
|
+
}
|
|
751
|
+
/*
|
|
752
|
+
A TOML table header, as in `[server]` or `[[servers]]` on a line of its own, reads as an array that holds a word, or as a key that starts with "[". Where a value is expected, `isValue`, it is only a table header when its bracket opens the document, because a table cannot be written as an item of an array inside it.
|
|
753
|
+
*/
|
|
754
|
+
#diagnoseTableHeader(isValue = false) {
|
|
755
|
+
const source = this.#source;
|
|
756
|
+
const lineStart = source.lastIndexOf('\n', this.#index - 1) + 1;
|
|
757
|
+
if (isValue && skipSpaces(source, lineStart) !== this.#documentStart) {
|
|
758
|
+
return;
|
|
544
759
|
}
|
|
545
|
-
|
|
760
|
+
const line = source.slice(lineStart, Math.min(findLineEnd(source, this.#index), lineStart + MAX_DIAGNOSED_LENGTH));
|
|
761
|
+
// The name is a valid key, so that the suggestions are valid: a dot has a segment on each side.
|
|
762
|
+
const match = /^[\t ]*\[\[?(?<name>[A-Z_a-z][\w\-]*(?:\.[\w\-]+)*)\]\]?[\t ]*$/v.exec(line);
|
|
763
|
+
if (match === null) {
|
|
764
|
+
return;
|
|
765
|
+
}
|
|
766
|
+
const { name } = match.groups;
|
|
767
|
+
this.#fail(`There are no table headers. Write the table as an object, as in ${abbreviate(name)}: {…}, or with dotted keys, as in ${abbreviate(name)}.key: …`);
|
|
768
|
+
}
|
|
769
|
+
/*
|
|
770
|
+
A YAML literal block scalar indicator, as in `key: |` or `key: |-`, at the end of its line.
|
|
771
|
+
*/
|
|
772
|
+
#isBlockScalarIndicator() {
|
|
773
|
+
const code = this.#code();
|
|
774
|
+
// A folded block scalar, `>`, joins its lines, which a block string does not, so only `|` gets the hint.
|
|
775
|
+
if (code !== 0x7C /* | */) {
|
|
776
|
+
return false;
|
|
777
|
+
}
|
|
778
|
+
const chomping = this.#code(this.#index + 1);
|
|
779
|
+
const end = skipSpaces(this.#source, chomping === DASH || chomping === 0x2B /* + */ ? this.#index + 2 : this.#index + 1);
|
|
780
|
+
const next = this.#code(end);
|
|
781
|
+
return next === LF || Number.isNaN(next);
|
|
782
|
+
}
|
|
783
|
+
/*
|
|
784
|
+
The hint for a `//` comment, as in JavaScript.
|
|
785
|
+
*/
|
|
786
|
+
#slashCommentHint() {
|
|
787
|
+
return this.#code() === SLASH && this.#code(this.#index + 1) === SLASH ? '. A comment starts with “#”' : '';
|
|
546
788
|
}
|
|
547
789
|
#describeHere() {
|
|
548
790
|
if (this.#isAtEnd()) {
|
|
@@ -602,8 +844,9 @@ class Parser {
|
|
|
602
844
|
#parseEscape(offset) {
|
|
603
845
|
const source = this.#source;
|
|
604
846
|
const character = source[offset + 1];
|
|
605
|
-
|
|
606
|
-
|
|
847
|
+
const simple = SIMPLE_ESCAPES.get(character ?? '');
|
|
848
|
+
if (simple !== undefined) {
|
|
849
|
+
return { text: simple, end: offset + 2 };
|
|
607
850
|
}
|
|
608
851
|
if (character === 'u') {
|
|
609
852
|
UNICODE_ESCAPE.lastIndex = offset + 1;
|
|
@@ -612,12 +855,10 @@ class Parser {
|
|
|
612
855
|
this.#fail(describeBadUnicodeEscape(source, offset + 1), offset);
|
|
613
856
|
}
|
|
614
857
|
const { hex } = match.groups;
|
|
615
|
-
if (hex.length > 1 && hex.startsWith('0')) {
|
|
616
|
-
this.#fail(String.raw `A Unicode escape may not have leading zeros; write \u{${hex.replace(/^0+(?=.)/v, '')}}`, offset);
|
|
617
|
-
}
|
|
618
858
|
const codePoint = Number.parseInt(hex, 16);
|
|
859
|
+
// The value is checked before the spelling, so that the leading zeros error never suggests an escape that is not allowed either.
|
|
619
860
|
if (codePoint === 0x0D) {
|
|
620
|
-
this.#fail(String.raw `A carriage return (U+000D) cannot be represented, so \u{
|
|
861
|
+
this.#fail(String.raw `A carriage return (U+000D) cannot be represented, so \u{${hex}} is not allowed`, offset);
|
|
621
862
|
}
|
|
622
863
|
if (codePoint >= 0xD8_00 && codePoint <= 0xDF_FF) {
|
|
623
864
|
this.#fail(String.raw `\u{${hex}} is a surrogate, which is not a Unicode scalar value`, offset);
|
|
@@ -625,10 +866,13 @@ class Parser {
|
|
|
625
866
|
if (codePoint > 0x10_FF_FF) {
|
|
626
867
|
this.#fail(String.raw `\u{${hex}} is above U+10FFFF, the largest Unicode scalar value`, offset);
|
|
627
868
|
}
|
|
869
|
+
if (hex.length > 1 && hex.startsWith('0')) {
|
|
870
|
+
this.#fail(String.raw `A Unicode escape may not have leading zeros; write \u{${hex.replace(/^0+(?=.)/v, '')}}`, offset);
|
|
871
|
+
}
|
|
628
872
|
return { text: String.fromCodePoint(codePoint), end: offset + 1 + match[0].length };
|
|
629
873
|
}
|
|
630
874
|
// A backslash at the end of the document is the same mistake as one at the end of a line.
|
|
631
|
-
const reason = ESCAPE_MISTAKES.get(character ?? '\n') ?? `Unknown escape
|
|
875
|
+
const reason = ESCAPE_MISTAKES.get(character ?? '\n') ?? `Unknown escape “\\${String.fromCodePoint(source.codePointAt(offset + 1))}”. The escapes are \\\\, \\", \\n, \\t, and \\u{…}; use a '...' string for literal backslashes`;
|
|
632
876
|
this.#fail(reason, offset);
|
|
633
877
|
}
|
|
634
878
|
/*
|
|
@@ -652,7 +896,7 @@ class Parser {
|
|
|
652
896
|
const contentStart = this.#index + 1;
|
|
653
897
|
const closing = findBlockStringEnd(source, contentStart, quote, delimiterLength);
|
|
654
898
|
if (closing === undefined) {
|
|
655
|
-
this.#fail(
|
|
899
|
+
this.#fail(`Unterminated block string${this.#describeInlineClosingDelimiter(contentStart, quote, delimiterLength)}`, start);
|
|
656
900
|
}
|
|
657
901
|
const indentation = source.slice(closing.lineStart, closing.delimiterStart);
|
|
658
902
|
this.#index = closing.delimiterStart + delimiterLength;
|
|
@@ -660,7 +904,7 @@ class Parser {
|
|
|
660
904
|
for (let lineStart = contentStart; lineStart < closing.lineStart; lineStart = findLineEnd(source, lineStart) + 1) {
|
|
661
905
|
const line = source.slice(lineStart, findLineEnd(source, lineStart));
|
|
662
906
|
// A blank line, which is empty or holds only spaces and tabs, may leave out the indentation, and it becomes an empty line. Swift is stricter here: there only a completely empty line may leave it out.
|
|
663
|
-
if (
|
|
907
|
+
if (isBlankLine(line)) {
|
|
664
908
|
contents.push({ text: '', offset: lineStart });
|
|
665
909
|
}
|
|
666
910
|
else if (line.startsWith(indentation)) {
|
|
@@ -682,6 +926,24 @@ class Parser {
|
|
|
682
926
|
const kept = contents.slice(first, last);
|
|
683
927
|
return quote === SINGLE_QUOTE ? kept.map(line => line.text).join('\n') : kept.map(line => this.#unescapeLine(line.text, line.offset)).join('\n');
|
|
684
928
|
}
|
|
929
|
+
/*
|
|
930
|
+
The hint for a content line that ends with the delimiter, as TOML allows. Adding a closing line after it would keep the delimiter as content.
|
|
931
|
+
*/
|
|
932
|
+
#describeInlineClosingDelimiter(contentStart, quote, delimiterLength) {
|
|
933
|
+
const source = this.#source;
|
|
934
|
+
const quoteCharacter = String.fromCharCode(quote);
|
|
935
|
+
const delimiter = quoteCharacter.repeat(delimiterLength);
|
|
936
|
+
let line = source.slice(0, contentStart).split('\n').length;
|
|
937
|
+
for (let lineStart = contentStart; lineStart < source.length; lineStart = findLineEnd(source, lineStart) + 1) {
|
|
938
|
+
const text = source.slice(skipSpaces(source, lineStart), skipSpacesBack(source, findLineEnd(source, lineStart)));
|
|
939
|
+
// A line of only quotes, longer than the delimiter, is content.
|
|
940
|
+
if (text.endsWith(delimiter) && text !== quoteCharacter.repeat(text.length)) {
|
|
941
|
+
return `. Its closing delimiter must start a line, so move the ${delimiter} at the end of line ${line} to a new line`;
|
|
942
|
+
}
|
|
943
|
+
line++;
|
|
944
|
+
}
|
|
945
|
+
return '';
|
|
946
|
+
}
|
|
685
947
|
#unescapeLine(text, offset) {
|
|
686
948
|
let value = '';
|
|
687
949
|
let chunkStart = 0;
|
|
@@ -710,17 +972,16 @@ class Parser {
|
|
|
710
972
|
return this.#parseDuration(text, start);
|
|
711
973
|
}
|
|
712
974
|
switch (kind) {
|
|
713
|
-
case '
|
|
714
|
-
case 'radix': {
|
|
975
|
+
case 'int': {
|
|
715
976
|
if (text === '-0') {
|
|
716
|
-
this.#fail('
|
|
977
|
+
this.#fail('“-0” is not allowed, because zero has one spelling: 0', start);
|
|
717
978
|
}
|
|
718
979
|
return this.#integer(text.replaceAll('_', ''), text, start);
|
|
719
980
|
}
|
|
720
981
|
case 'float': {
|
|
721
982
|
const value = Number(text.replaceAll('_', ''));
|
|
722
983
|
if (!Number.isFinite(value)) {
|
|
723
|
-
this.#fail(`${abbreviate(text)} is too large to be a finite float. Use infinity if you mean it`, start);
|
|
984
|
+
this.#fail(`${abbreviate(text)} is too large to be a finite float. Use ${value < 0 ? '-infinity' : 'infinity'} if you mean it`, start);
|
|
724
985
|
}
|
|
725
986
|
if (value === 0) {
|
|
726
987
|
// A nonzero digit before the exponent means the literal is not zero, so it underflowed.
|
|
@@ -736,7 +997,49 @@ class Parser {
|
|
|
736
997
|
break;
|
|
737
998
|
}
|
|
738
999
|
}
|
|
739
|
-
this.#fail(describeBadNumber(text), start);
|
|
1000
|
+
this.#fail(this.#describeThousandsSeparator(text, start) ?? describeBadNumber(text, this.#unquotedText(start)), start);
|
|
1001
|
+
}
|
|
1002
|
+
/*
|
|
1003
|
+
A comma used as a thousands separator, as in `[1,000]`, makes the group after it a separate item, which fails only when it has a leading zero. Writing that item in octal, as the leading zero message suggests, would parse to the wrong items.
|
|
1004
|
+
*/
|
|
1005
|
+
#describeThousandsSeparator(text, start) {
|
|
1006
|
+
if (!/^0\d{2}$/v.test(text) || this.#code(start - 1) !== COMMA) {
|
|
1007
|
+
return undefined;
|
|
1008
|
+
}
|
|
1009
|
+
const getGroupStart = (groupEnd) => {
|
|
1010
|
+
let groupStart = groupEnd;
|
|
1011
|
+
while (isDigit(this.#code(groupStart - 1))) {
|
|
1012
|
+
groupStart--;
|
|
1013
|
+
}
|
|
1014
|
+
return groupStart;
|
|
1015
|
+
};
|
|
1016
|
+
let groupEnd = start - 1;
|
|
1017
|
+
let numberStart = getGroupStart(groupEnd);
|
|
1018
|
+
// Earlier groups of three digits, as in `12,345,000`.
|
|
1019
|
+
while (groupEnd - numberStart === 3 && this.#code(numberStart - 1) === COMMA && isDigit(this.#code(numberStart - 2))) {
|
|
1020
|
+
groupEnd = numberStart - 1;
|
|
1021
|
+
numberStart = getGroupStart(groupEnd);
|
|
1022
|
+
}
|
|
1023
|
+
const before = this.#code(numberStart - 1);
|
|
1024
|
+
// The first group has one to three digits, which are not the end of a float, of a number with underscores, or of a number with a radix prefix.
|
|
1025
|
+
if (numberStart === groupEnd
|
|
1026
|
+
|| before === DOT
|
|
1027
|
+
|| before === 0x5F /* _ */
|
|
1028
|
+
|| numberStart < groupEnd - 3
|
|
1029
|
+
|| isLetter(before)) {
|
|
1030
|
+
return undefined;
|
|
1031
|
+
}
|
|
1032
|
+
// The sign belongs to the number, as in `-1,000`.
|
|
1033
|
+
if (before === DASH) {
|
|
1034
|
+
numberStart--;
|
|
1035
|
+
}
|
|
1036
|
+
const groupsStart = start + text.length;
|
|
1037
|
+
const groups = /^(?:,\d{3}(?!\d))*/v.exec(this.#source.slice(groupsStart, groupsStart + MAX_DIAGNOSED_LENGTH))[0];
|
|
1038
|
+
const number = this.#source.slice(numberStart, groupsStart + groups.length);
|
|
1039
|
+
const reason = `Leading zeros are not allowed in a decimal number. A comma separates items, so ${abbreviate(number)} is not one number`;
|
|
1040
|
+
// Both spellings are the same int, so they are suggested only when it is in range.
|
|
1041
|
+
const withUnderscores = number.replaceAll(',', '_');
|
|
1042
|
+
return isValidValue(withUnderscores) ? `${reason}. Write it as ${abbreviate(withUnderscores)} or ${abbreviate(number.replaceAll(',', ''))}` : reason;
|
|
740
1043
|
}
|
|
741
1044
|
/*
|
|
742
1045
|
The common case, a short decimal int or float such as `8080`, `-3`, or `30.5`, without the general path's maximal-run scan and classification. Anything else, including every error, returns `undefined` and is left to the general path.
|
|
@@ -750,7 +1053,7 @@ class Parser {
|
|
|
750
1053
|
index++;
|
|
751
1054
|
}
|
|
752
1055
|
const integerLength = index - integerStart;
|
|
753
|
-
// No digits, or a leading zero, which
|
|
1056
|
+
// No digits, or a leading zero followed by more digits, which the general path reports as an error or reads as an instant before the year 1000.
|
|
754
1057
|
if (integerLength === 0 || (integerLength > 1 && source.charCodeAt(integerStart) === 0x30)) {
|
|
755
1058
|
return;
|
|
756
1059
|
}
|
|
@@ -766,9 +1069,8 @@ class Parser {
|
|
|
766
1069
|
}
|
|
767
1070
|
isFloat = true;
|
|
768
1071
|
}
|
|
769
|
-
const next = source.charCodeAt(index);
|
|
770
1072
|
// At most 15 digits, so an int is exact as a number. A character that may continue a number, such as `e`, `_`, or `-`, needs the general path.
|
|
771
|
-
if (index - integerStart > 15 || (
|
|
1073
|
+
if (index - integerStart > 15 || !isValueEnd(source.charCodeAt(index))) {
|
|
772
1074
|
return;
|
|
773
1075
|
}
|
|
774
1076
|
const value = Number(source.slice(start, index));
|
|
@@ -811,7 +1113,14 @@ class Parser {
|
|
|
811
1113
|
#parseInstant(text, start) {
|
|
812
1114
|
const match = text.length > MAX_DIAGNOSED_LENGTH ? null : INSTANT.exec(text);
|
|
813
1115
|
if (match === null) {
|
|
814
|
-
|
|
1116
|
+
// A date, a space, and a time, as TOML and Python write a date and time.
|
|
1117
|
+
const timeStart = this.#index + 1;
|
|
1118
|
+
const end = this.#code() === SPACE && isDigit(this.#code(timeStart)) ? findNumberEnd(this.#source, timeStart) : this.#index;
|
|
1119
|
+
const time = this.#source.slice(timeStart, end);
|
|
1120
|
+
// More of the instant may follow, as in `14:00:00,5Z` or `14:00:00[Europe/Oslo]`, so it may have an offset.
|
|
1121
|
+
const next = this.#code(end);
|
|
1122
|
+
const isWholeValue = isValueEnd(next) && !(next === COMMA && isDigit(this.#code(end + 1)));
|
|
1123
|
+
this.#fail(describeBadInstant(text, time, isWholeValue), start);
|
|
815
1124
|
}
|
|
816
1125
|
const { year, month, day, hour, minute, second, fraction, offset, offsetHour, offsetMinute } = match.groups;
|
|
817
1126
|
const check = (isValid, reason) => {
|
|
@@ -830,12 +1139,13 @@ class Parser {
|
|
|
830
1139
|
if (offset !== 'Z') {
|
|
831
1140
|
check(offsetHour <= '23', 'the offset hour must be 00 to 23');
|
|
832
1141
|
check(offsetMinute <= '59', 'the offset minute must be 00 to 59');
|
|
833
|
-
check(offset !== '-00:00', '-00:00 means
|
|
1142
|
+
check(offset !== '-00:00', '-00:00 means “offset unknown” in RFC 3339, which is not representable; use Z or +00:00');
|
|
834
1143
|
}
|
|
835
|
-
|
|
836
|
-
const
|
|
1144
|
+
// `Date.parse()` reads this format exactly, for every year from 0001 to 9999 and every offset, so the range check needs no `Temporal`.
|
|
1145
|
+
const milliseconds = Date.parse(`${year}-${month}-${day}T${hour}:${minute}:${second}${offset}`);
|
|
1146
|
+
const nanoseconds = (BigInt(milliseconds) * 1000000n) + BigInt((fraction ?? '').padEnd(9, '0'));
|
|
837
1147
|
check(nanoseconds >= MIN_INSTANT && nanoseconds <= MAX_INSTANT, 'in UTC it falls outside the years 0001 to 9999');
|
|
838
|
-
return
|
|
1148
|
+
return this.#createTime(new Time('Instant', nanoseconds));
|
|
839
1149
|
}
|
|
840
1150
|
/*
|
|
841
1151
|
A scan of the parts in order, so that each error names the part it is about.
|
|
@@ -859,7 +1169,7 @@ class Parser {
|
|
|
859
1169
|
if (code === DASH && isDigit(text.charCodeAt(index + 1))) {
|
|
860
1170
|
fail('only the whole duration takes a sign, as in -1h30m');
|
|
861
1171
|
}
|
|
862
|
-
fail(code === 0x2B /* + */ ? 'a
|
|
1172
|
+
fail(code === 0x2B /* + */ ? 'a “+” sign is not allowed' : `expected a number at “${abbreviate(text.slice(index), 10)}”`);
|
|
863
1173
|
}
|
|
864
1174
|
if (text.charCodeAt(index) === 0x30 && (next === 0x5F /* _ */ || isDigit(next))) {
|
|
865
1175
|
fail('leading zeros are not allowed');
|
|
@@ -872,7 +1182,7 @@ class Parser {
|
|
|
872
1182
|
if (next === DOT) {
|
|
873
1183
|
unitStart = skipDigits(text, digitsEnd + 1, isDigit);
|
|
874
1184
|
if (unitStart === digitsEnd + 1) {
|
|
875
|
-
fail('a
|
|
1185
|
+
fail('a “.” must be followed by a digit');
|
|
876
1186
|
}
|
|
877
1187
|
if (text.charCodeAt(unitStart) === 0x5F /* _ */) {
|
|
878
1188
|
fail('an underscore must be between two digits');
|
|
@@ -887,7 +1197,7 @@ class Parser {
|
|
|
887
1197
|
const unit = text.slice(unitStart, unitEnd);
|
|
888
1198
|
const rank = DURATION_UNIT_NAMES.indexOf(unit);
|
|
889
1199
|
if (rank === -1) {
|
|
890
|
-
fail(describeBadDurationUnit(unit));
|
|
1200
|
+
fail(describeBadDurationUnit(unit, text));
|
|
891
1201
|
}
|
|
892
1202
|
if (rank <= previousRank) {
|
|
893
1203
|
fail('the units are in the order h, m, s, ms, us, ns, and each appears at most once');
|
|
@@ -915,15 +1225,16 @@ class Parser {
|
|
|
915
1225
|
total += fractionNanoseconds / scale;
|
|
916
1226
|
}
|
|
917
1227
|
if (isNegative && total === 0n) {
|
|
918
|
-
fail('
|
|
1228
|
+
fail('“-” is not allowed before zero, because zero has one spelling: 0s');
|
|
919
1229
|
}
|
|
920
1230
|
if (total > (isNegative ? -INT64_MIN : INT64_MAX)) {
|
|
921
1231
|
fail('it is outside the 64-bit range of nanoseconds, about 292 years either way');
|
|
922
1232
|
}
|
|
923
|
-
return
|
|
1233
|
+
return this.#createTime(new Time('Duration', isNegative ? -total : total));
|
|
924
1234
|
}
|
|
925
1235
|
parseDocument() {
|
|
926
1236
|
this.#skipTrivia();
|
|
1237
|
+
this.#documentStart = this.#index;
|
|
927
1238
|
if (this.#isAtEnd()) {
|
|
928
1239
|
this.#fail('A document must contain an object or an array, but this one is empty');
|
|
929
1240
|
}
|
|
@@ -946,13 +1257,24 @@ class Parser {
|
|
|
946
1257
|
}
|
|
947
1258
|
}
|
|
948
1259
|
/*
|
|
949
|
-
Whether `text` is
|
|
1260
|
+
Whether `text` is one valid value, so that an error message only suggests a fix that works.
|
|
1261
|
+
*/
|
|
1262
|
+
function isValidValue(text) {
|
|
1263
|
+
try {
|
|
1264
|
+
return new Parser(`[${text}]`, 'bigint', time => time).parseDocument().length === 1;
|
|
1265
|
+
}
|
|
1266
|
+
catch {
|
|
1267
|
+
return false;
|
|
1268
|
+
}
|
|
1269
|
+
}
|
|
1270
|
+
/*
|
|
1271
|
+
Whether `text` is an int, in any radix, or a float, following the grammar. A run of digits may have single underscores between digits.
|
|
950
1272
|
*/
|
|
951
1273
|
function classifyNumber(text) {
|
|
952
|
-
const radix = text.charCodeAt(0) === 0x30 ? RADIX_DIGIT
|
|
1274
|
+
const radix = text.charCodeAt(0) === 0x30 ? RADIX_DIGIT.get(text.charAt(1)) : undefined;
|
|
953
1275
|
if (radix !== undefined) {
|
|
954
1276
|
const end = skipDigits(text, 2, radix);
|
|
955
|
-
return end > 2 && end === text.length ? '
|
|
1277
|
+
return end > 2 && end === text.length ? 'int' : undefined;
|
|
956
1278
|
}
|
|
957
1279
|
let index = text.charCodeAt(0) === DASH ? 1 : 0;
|
|
958
1280
|
// The integer part is `0` or has no leading zero.
|
|
@@ -961,7 +1283,7 @@ function classifyNumber(text) {
|
|
|
961
1283
|
return undefined;
|
|
962
1284
|
}
|
|
963
1285
|
index = integerEnd;
|
|
964
|
-
let kind = '
|
|
1286
|
+
let kind = 'int';
|
|
965
1287
|
if (text.charCodeAt(index) === DOT) {
|
|
966
1288
|
const end = skipDigits(text, index + 1, isDigit);
|
|
967
1289
|
if (end === index + 1) {
|
|
@@ -1037,7 +1359,7 @@ function isDurationLike(text) {
|
|
|
1037
1359
|
}
|
|
1038
1360
|
return /^[dhmnsuw]$/iv.test(text.charAt(index));
|
|
1039
1361
|
}
|
|
1040
|
-
function describeBadDurationUnit(unit) {
|
|
1362
|
+
function describeBadDurationUnit(unit, text) {
|
|
1041
1363
|
if (unit === '') {
|
|
1042
1364
|
return 'every number needs a unit: h, m, s, ms, us, or ns';
|
|
1043
1365
|
}
|
|
@@ -1047,12 +1369,24 @@ function describeBadDurationUnit(unit) {
|
|
|
1047
1369
|
if (/^w(?:eeks?)?$/v.test(unit)) {
|
|
1048
1370
|
return 'there is no week unit, because a day is not a fixed length. Write 168h for a fixed 168 hours';
|
|
1049
1371
|
}
|
|
1050
|
-
|
|
1372
|
+
if (unit === 'M') {
|
|
1373
|
+
// A number with `M` alone is more often a size, as in `memory: 512M`, and `512m` would read as minutes.
|
|
1374
|
+
const sizeHint = text.length <= MAX_DIAGNOSED_LENGTH && /^\d[\d_]*M$/v.test(text) ? `. A size, such as 512M, is a string: '${abbreviate(text)}'` : '';
|
|
1375
|
+
return `there is no month unit, because a month is not a fixed length${sizeHint}`;
|
|
1376
|
+
}
|
|
1377
|
+
return DURATION_UNITS.has(unit.toLowerCase()) ? `the units are lowercase: ${unit.toLowerCase()}` : `“${abbreviate(unit, 10)}” is not a unit. The units are h, m, s, ms, us, and ns, and a string must be quoted`;
|
|
1378
|
+
}
|
|
1379
|
+
/*
|
|
1380
|
+
The number of spaces and tabs before `offset` on its line, or `undefined` when something else comes before it.
|
|
1381
|
+
*/
|
|
1382
|
+
function getIndentation(source, offset) {
|
|
1383
|
+
const lineStart = source.lastIndexOf('\n', offset - 1) + 1;
|
|
1384
|
+
return skipSpaces(source, lineStart) === offset ? offset - lineStart : undefined;
|
|
1051
1385
|
}
|
|
1052
1386
|
function isWordsWithSpaces(text) {
|
|
1053
1387
|
for (let index = 0; index < text.length; index++) {
|
|
1054
1388
|
const code = text.charCodeAt(index);
|
|
1055
|
-
if (code
|
|
1389
|
+
if (!isSpace(code) && !isBareKeyCharacter(code)) {
|
|
1056
1390
|
return false;
|
|
1057
1391
|
}
|
|
1058
1392
|
}
|
|
@@ -1065,23 +1399,38 @@ function daysInMonth(year, month) {
|
|
|
1065
1399
|
}
|
|
1066
1400
|
return [4, 6, 9, 11].includes(month) ? 30 : 31;
|
|
1067
1401
|
}
|
|
1068
|
-
function
|
|
1069
|
-
const
|
|
1402
|
+
function isUnquotedTextEnd(source, index) {
|
|
1403
|
+
const code = source.charCodeAt(index);
|
|
1404
|
+
const isComment = code === HASH || (code === SLASH && source.charCodeAt(index + 1) === ASTERISK);
|
|
1405
|
+
return code === LF || code === COMMA || code === CLOSE_BRACKET || code === CLOSE_BRACE || (isComment && isSpace(source.charCodeAt(index - 1)));
|
|
1406
|
+
}
|
|
1407
|
+
/*
|
|
1408
|
+
The example in a message that a string must be quoted, which quotes `text` as a `'...'` string. That cannot hold a `'`, so then there is no simple example.
|
|
1409
|
+
*/
|
|
1410
|
+
function quotingExample(text) {
|
|
1411
|
+
return text.includes('\'') ? '' : `, as in '${abbreviate(text)}'`;
|
|
1412
|
+
}
|
|
1413
|
+
/*
|
|
1414
|
+
`text` is the unquoted string that `word` begins, which the suggestion quotes.
|
|
1415
|
+
*/
|
|
1416
|
+
function describeUnknownWord(word, text) {
|
|
1417
|
+
// A keyword hint is only for the word on its own. In `Yes please` or `Nan Goldin`, following it would leave text behind or change the value.
|
|
1418
|
+
const lowercase = text === word ? word.toLowerCase() : '';
|
|
1070
1419
|
if (['true', 'false', 'yes', 'no', 'on', 'off'].includes(lowercase)) {
|
|
1071
|
-
return
|
|
1420
|
+
return `“${word}” is not a value. Booleans are written true and false, in lowercase`;
|
|
1072
1421
|
}
|
|
1073
1422
|
if (['null', 'nil', 'none', 'undefined'].includes(lowercase)) {
|
|
1074
|
-
return
|
|
1423
|
+
return `“${word}” is not a value. Null is written null, in lowercase`;
|
|
1075
1424
|
}
|
|
1076
1425
|
if (lowercase === 'nan') {
|
|
1077
1426
|
return 'NaN is not representable. Use null for a missing value';
|
|
1078
1427
|
}
|
|
1079
|
-
return ['inf', 'infinity'].includes(lowercase) ?
|
|
1428
|
+
return ['inf', 'infinity'].includes(lowercase) ? `“${word}” is not a value. Infinity is written infinity, in lowercase` : `Unexpected “${abbreviate(word)}”. A string value must be quoted${quotingExample(text)}`;
|
|
1080
1429
|
}
|
|
1081
1430
|
function describeBadUnicodeEscape(source, offset) {
|
|
1082
1431
|
const rest = source.slice(offset, offset + 12);
|
|
1083
1432
|
if (/^u[\dA-Fa-f]{4}/v.test(rest)) {
|
|
1084
|
-
return
|
|
1433
|
+
return describeFourDigitEscape(rest);
|
|
1085
1434
|
}
|
|
1086
1435
|
if (/^u\{[\dA-Fa-f]*[A-F]/v.test(rest)) {
|
|
1087
1436
|
return 'A Unicode escape uses lowercase hexadecimal digits';
|
|
@@ -1091,15 +1440,36 @@ function describeBadUnicodeEscape(source, offset) {
|
|
|
1091
1440
|
}
|
|
1092
1441
|
return /^u\{[\da-f]{7}/v.test(rest) ? 'A Unicode escape has at most six hexadecimal digits' : String.raw `A Unicode escape is written \u{…} with one to six lowercase hexadecimal digits`;
|
|
1093
1442
|
}
|
|
1094
|
-
|
|
1443
|
+
/*
|
|
1444
|
+
The JSON form `\uXXXX`, from its `u`, with the escape to write instead. JSON writes a character above U+FFFF as two of them, a surrogate pair, which is one `\u{…}` escape here. Here, a lone surrogate and a carriage return have no escape.
|
|
1445
|
+
*/
|
|
1446
|
+
function describeFourDigitEscape(rest) {
|
|
1447
|
+
const code = Number.parseInt(rest.slice(1, 5), 16);
|
|
1448
|
+
const low = /^\\u[\dA-Fa-f]{4}/v.test(rest.slice(5)) ? rest.slice(7, 11) : undefined;
|
|
1449
|
+
const lowCode = low === undefined ? NaN : Number.parseInt(low, 16);
|
|
1450
|
+
if (code >= 0xD8_00 && code <= 0xDB_FF && lowCode >= 0xDC_00 && lowCode <= 0xDF_FF) {
|
|
1451
|
+
const codePoint = 0x1_00_00 + ((code - 0xD8_00) * 0x4_00) + (lowCode - 0xDC_00);
|
|
1452
|
+
return `The four-digit \\${rest.slice(0, 5)}\\u${low} form is not an escape. Write \\u{${codePoint.toString(16)}}`;
|
|
1453
|
+
}
|
|
1454
|
+
const form = `The four-digit \\${rest.slice(0, 5)} form is not an escape`;
|
|
1455
|
+
if (code >= 0xD8_00 && code <= 0xDF_FF) {
|
|
1456
|
+
return String.raw `${form}, and a lone surrogate is not a Unicode scalar value. Write the character it is half of as one \u{…} escape`;
|
|
1457
|
+
}
|
|
1458
|
+
return code === 0x0D ? `${form}, and a carriage return (U+000D) cannot be represented` : String.raw `${form}. Write \u{${code.toString(16)}}`;
|
|
1459
|
+
}
|
|
1460
|
+
/*
|
|
1461
|
+
`unquotedText` is the unquoted string that `fullText` begins, for the suggestion to quote it.
|
|
1462
|
+
*/
|
|
1463
|
+
function describeBadNumber(fullText, unquotedText) {
|
|
1095
1464
|
const text = fullText.slice(0, MAX_DIAGNOSED_LENGTH);
|
|
1096
1465
|
if (text.includes('+')) {
|
|
1097
|
-
return 'A
|
|
1466
|
+
return 'A “+” sign is not allowed in a number, including in an exponent';
|
|
1098
1467
|
}
|
|
1099
1468
|
if (/^-?0[BOX]/v.test(text)) {
|
|
1100
1469
|
return 'A number prefix is lowercase: 0x, 0o, or 0b';
|
|
1101
1470
|
}
|
|
1102
|
-
|
|
1471
|
+
// The whole number, not the part cut at the diagnosed length, because a cut can end before the digit or the underscore that the message is about. Each check below takes linear time.
|
|
1472
|
+
const radixMatch = /^(?<sign>-?)0(?<radix>[box])(?<digits>.*)$/v.exec(fullText);
|
|
1103
1473
|
if (radixMatch !== null) {
|
|
1104
1474
|
const { sign, radix, digits } = radixMatch.groups;
|
|
1105
1475
|
const name = RADIX_NAME[radix];
|
|
@@ -1107,26 +1477,44 @@ function describeBadNumber(fullText) {
|
|
|
1107
1477
|
return `${RADIX_ARTICLE[radix]} ${name} integer cannot have a sign, because it states a bit pattern rather than a quantity`;
|
|
1108
1478
|
}
|
|
1109
1479
|
if (digits === '') {
|
|
1110
|
-
return `Expected ${name} digits after
|
|
1480
|
+
return `Expected ${name} digits after “0${radix}”`;
|
|
1111
1481
|
}
|
|
1112
1482
|
if (radix === 'x' && /[a-f]/v.test(digits) && !/[^\da-f_]/iv.test(digits)) {
|
|
1113
|
-
|
|
1483
|
+
// The uppercase spelling is only suggested when it is valid, so a misplaced underscore is reported first, and a value outside the 64-bit range gets no example.
|
|
1484
|
+
if (!/^[\da-f]+(?:_[\da-f]+)*$/iv.test(digits)) {
|
|
1485
|
+
return 'An underscore in a number must be between two digits';
|
|
1486
|
+
}
|
|
1487
|
+
const uppercase = `0x${digits.toUpperCase()}`;
|
|
1488
|
+
return isValidValue(uppercase) ? `Hexadecimal digits are uppercase: ${abbreviate(uppercase)}` : 'Hexadecimal digits are uppercase';
|
|
1114
1489
|
}
|
|
1115
|
-
|
|
1490
|
+
// A lowercase hexadecimal digit is a digit in the wrong case, so the one named is a character that is no digit in either case.
|
|
1491
|
+
const validCharacter = { x: /[\da-f_]/iv, o: /[0-7_]/v, b: /[01_]/v }[radix];
|
|
1116
1492
|
const character = [...digits].find(character => !validCharacter.test(character));
|
|
1117
|
-
return character === undefined ? 'An underscore in a number must be between two digits' : `Invalid ${name} digit
|
|
1493
|
+
return character === undefined ? 'An underscore in a number must be between two digits' : `Invalid ${name} digit “${character}”`;
|
|
1118
1494
|
}
|
|
1119
1495
|
if (/^-?\d[\d_]*E/v.test(text) || /^-?\d[\d_]*\.[\d_]+E/v.test(text)) {
|
|
1120
|
-
return 'An exponent marker is a lowercase
|
|
1496
|
+
return 'An exponent marker is a lowercase “e”';
|
|
1121
1497
|
}
|
|
1122
1498
|
if (/^-?[\d_]+\.(?:$|[^\d_])/v.test(text)) {
|
|
1123
1499
|
return 'A decimal point must be followed by a digit';
|
|
1124
1500
|
}
|
|
1125
1501
|
if (text.startsWith('-.')) {
|
|
1126
|
-
return 'A number cannot begin with
|
|
1502
|
+
return 'A number cannot begin with “.”; write a digit before it, as in -0.5';
|
|
1503
|
+
}
|
|
1504
|
+
if (/^\d+(?::\d+)+$/v.test(text)) {
|
|
1505
|
+
return /^\d{1,2}:\d{2}(?::\d{2}(?:\.\d+)?)?$/v.test(text) ? 'A time of day is a string, so it must be quoted' : `Invalid number “${abbreviate(text)}”. A value that contains “:” must be quoted, as in '${abbreviate(text)}'`;
|
|
1127
1506
|
}
|
|
1128
1507
|
if (/^-?0[\d_]/v.test(text)) {
|
|
1129
|
-
|
|
1508
|
+
// Removing the zero gives a valid number with a different meaning, so the message says what the zero usually meant.
|
|
1509
|
+
// A `'...'` string cannot hold a `'`, so then there is no example.
|
|
1510
|
+
const identifier = `an identifier, such as a ZIP code, as a string${unquotedText.includes('\'') ? '' : `: '${abbreviate(unquotedText)}'`}`;
|
|
1511
|
+
// The octal suggestion is only for the number on its own, not for one that more text follows, as in `0412 345 678`, and only when it is in range. It is decided on the whole number, because a cut can end before a digit that is not octal or that changes the value.
|
|
1512
|
+
const digits = fullText.replace(/^0+/v, '');
|
|
1513
|
+
const octal = `0o${digits === '' ? '0' : digits}`;
|
|
1514
|
+
if (unquotedText === fullText && /^0[0-7]+$/v.test(fullText) && isValidValue(octal)) {
|
|
1515
|
+
return `Leading zeros are not allowed in a decimal number. Write an octal number, such as a file mode, as ${abbreviate(octal)}, and ${identifier}`;
|
|
1516
|
+
}
|
|
1517
|
+
return /^0\d+$/v.test(text) ? `Leading zeros are not allowed in a decimal number. Write ${identifier}` : 'Leading zeros are not allowed in a decimal number';
|
|
1130
1518
|
}
|
|
1131
1519
|
if (/_(?:$|\D)|(?:^|\D)_/v.test(text)) {
|
|
1132
1520
|
return 'An underscore in a number must be between two digits';
|
|
@@ -1135,37 +1523,65 @@ function describeBadNumber(fullText) {
|
|
|
1135
1523
|
return 'Leading zeros are not allowed in an exponent';
|
|
1136
1524
|
}
|
|
1137
1525
|
if (/^-?\d[\d_]*(?:\.[\d_]+)?e-0$/v.test(text)) {
|
|
1138
|
-
return '
|
|
1526
|
+
return '“e-0” is not allowed, because an exponent of zero has one spelling: e0';
|
|
1139
1527
|
}
|
|
1140
1528
|
if (/e-?$/v.test(text)) {
|
|
1141
|
-
return 'Expected digits after the exponent marker
|
|
1529
|
+
return 'Expected digits after the exponent marker “e”';
|
|
1142
1530
|
}
|
|
1143
1531
|
if (text.split('.').length > 2) {
|
|
1144
|
-
return `Invalid number
|
|
1532
|
+
return `Invalid number “${abbreviate(text)}”. A value with several dots, such as a version number, must be quoted`;
|
|
1145
1533
|
}
|
|
1146
1534
|
if (/^-(?:\D|$)/v.test(text)) {
|
|
1147
1535
|
if (/^-nan/iv.test(text)) {
|
|
1148
1536
|
return 'NaN is not representable. Use null for a missing value';
|
|
1149
1537
|
}
|
|
1150
|
-
return /^-inf/iv.test(text) ?
|
|
1538
|
+
return /^-inf/iv.test(text) ? `“${abbreviate(text)}” is not a value. Negative infinity is written -infinity` : 'Expected a digit or “infinity” after “-”';
|
|
1151
1539
|
}
|
|
1152
|
-
return /[A-Za-z]/v.test(text) ? `Invalid number
|
|
1540
|
+
return /[A-Za-z]/v.test(text) ? `Invalid number “${abbreviate(text)}”. A string value must be quoted${quotingExample(unquotedText)}` : `Invalid number “${abbreviate(text)}”`;
|
|
1153
1541
|
}
|
|
1154
|
-
|
|
1542
|
+
/*
|
|
1543
|
+
A date alone, which is not an instant. The instant it could be is only shown when the date exists, so that the example is valid.
|
|
1544
|
+
*/
|
|
1545
|
+
function describeDate(date) {
|
|
1546
|
+
const [year, month, day] = date.split('-').map(Number);
|
|
1547
|
+
const isExisting = year >= 1 && month >= 1 && month <= 12 && day >= 1 && day <= daysInMonth(year, month);
|
|
1548
|
+
return `${date} is a date, not an instant. Write a date as a string, as in '${date}'${isExisting ? `. An instant needs a time and an offset, as in ${date}T00:00:00Z` : ''}`;
|
|
1549
|
+
}
|
|
1550
|
+
/*
|
|
1551
|
+
`time` is the token after a space that follows `text`, or an empty string. `isWholeValue` is whether nothing that may be part of the instant follows `text`.
|
|
1552
|
+
*/
|
|
1553
|
+
function describeBadInstant(text, time, isWholeValue) {
|
|
1155
1554
|
if (text.length > MAX_DIAGNOSED_LENGTH) {
|
|
1156
|
-
return `Invalid instant
|
|
1555
|
+
return `Invalid instant “${abbreviate(text)}”. ${INSTANT_FORMAT}`;
|
|
1157
1556
|
}
|
|
1158
1557
|
if (/^\d{4}-\d{2}-\d{2}$/v.test(text)) {
|
|
1159
|
-
return
|
|
1558
|
+
return /^\d{2}:\d{2}/v.test(time) ? describeSpaceSeparatedInstant(text, time, isWholeValue) : describeDate(text);
|
|
1160
1559
|
}
|
|
1161
1560
|
if (/^\d{4}-\d{2}-\d{2}t/v.test(text)) {
|
|
1162
|
-
return 'The date and time separator in an instant is an uppercase
|
|
1561
|
+
return 'The date and time separator in an instant is an uppercase “T”';
|
|
1163
1562
|
}
|
|
1164
1563
|
if (text.endsWith('z')) {
|
|
1165
|
-
return 'The UTC offset in an instant is an uppercase
|
|
1564
|
+
return 'The UTC offset in an instant is an uppercase “Z”';
|
|
1166
1565
|
}
|
|
1167
1566
|
if (/[+\-]\d{4}$/v.test(text)) {
|
|
1168
1567
|
return 'An instant\'s offset is written with a colon, as in +07:00';
|
|
1169
1568
|
}
|
|
1170
|
-
|
|
1569
|
+
if (!LOCAL_DATE_TIME.test(text)) {
|
|
1570
|
+
return `Invalid instant “${abbreviate(text)}”. ${INSTANT_FORMAT}`;
|
|
1571
|
+
}
|
|
1572
|
+
return isWholeValue ? `An instant needs an offset: Z or ±HH:MM. A date and time without an offset is a local time, which is not an instant, so write it as a string, as in '${abbreviate(text)}', or add the offset it was meant in` : 'An instant needs an offset: Z or ±HH:MM';
|
|
1573
|
+
}
|
|
1574
|
+
/*
|
|
1575
|
+
A date and a time with a space between them, where an instant has a "T". Following the example must not turn a local time into UTC silently, so a time without an offset is described as what it is. An instant is only shown when it is valid, so a date that does not exist, a time out of range, or an instant outside the years 0001 to 9999 in UTC gets the general format instead.
|
|
1576
|
+
*/
|
|
1577
|
+
function describeSpaceSeparatedInstant(date, time, isWholeValue) {
|
|
1578
|
+
const separator = 'The date and time separator in an instant is an uppercase “T”, not a space';
|
|
1579
|
+
const instant = `${date}T${time}`;
|
|
1580
|
+
if (instant.length > MAX_DIAGNOSED_LENGTH) {
|
|
1581
|
+
return `${separator}. ${INSTANT_FORMAT}`;
|
|
1582
|
+
}
|
|
1583
|
+
if (isValidValue(instant)) {
|
|
1584
|
+
return `${separator}, as in ${abbreviate(instant)}`;
|
|
1585
|
+
}
|
|
1586
|
+
return isWholeValue && LOCAL_DATE_TIME.test(instant) && isValidValue(`${instant}Z`) ? `${separator}, and an instant needs the offset it was meant in, as in ${abbreviate(instant)}Z for UTC. A date and time without an offset is a local time, which is not an instant, so write it as a string, as in '${abbreviate(`${date} ${time}`)}'` : `${separator}. ${INSTANT_FORMAT}`;
|
|
1171
1587
|
}
|