soml-lang 0.0.2 → 0.0.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/distribution/edit.d.ts +56 -0
- package/distribution/edit.js +536 -0
- package/distribution/error.d.ts +19 -2
- package/distribution/error.js +20 -3
- package/distribution/format.d.ts +48 -4
- package/distribution/format.js +209 -49
- package/distribution/index.d.ts +6 -3
- package/distribution/index.js +7 -2
- package/distribution/parse.d.ts +69 -6
- package/distribution/parse.js +603 -227
- package/distribution/shared.d.ts +58 -4
- package/distribution/shared.js +93 -11
- package/distribution/stringify.d.ts +88 -8
- package/distribution/stringify.js +173 -59
- package/distribution/tree.d.ts +111 -9
- package/distribution/tree.js +214 -95
- package/package.json +4 -3
- package/readme.md +256 -50
package/distribution/parse.js
CHANGED
|
@@ -1,43 +1,41 @@
|
|
|
1
1
|
import { ParseError } from "./error.js";
|
|
2
|
-
import {
|
|
3
|
-
|
|
4
|
-
const LF = 0x0A;
|
|
5
|
-
const SPACE = 0x20;
|
|
6
|
-
const DOUBLE_QUOTE = 0x22;
|
|
7
|
-
const HASH = 0x23;
|
|
8
|
-
const SINGLE_QUOTE = 0x27;
|
|
9
|
-
const ASTERISK = 0x2A;
|
|
10
|
-
const COMMA = 0x2C;
|
|
11
|
-
const DASH = 0x2D;
|
|
12
|
-
const DOT = 0x2E;
|
|
13
|
-
const SLASH = 0x2F;
|
|
14
|
-
const COLON = 0x3A;
|
|
15
|
-
const OPEN_BRACKET = 0x5B;
|
|
16
|
-
const CLOSE_BRACKET = 0x5D;
|
|
17
|
-
const OPEN_BRACE = 0x7B;
|
|
18
|
-
const CLOSE_BRACE = 0x7D;
|
|
19
|
-
/*
|
|
20
|
-
What may directly follow a scalar value: whitespace, a separator, a closing bracket, a comment, or the end.
|
|
21
|
-
*/
|
|
2
|
+
import { formatKey } from "./stringify.js";
|
|
3
|
+
import { MAX_DEPTH, INT64_MIN, INT64_MAX, MIN_INSTANT, MAX_INSTANT, DURATION_UNITS, createDuration, requireTemporal, trimTrailingZeros, getIntegersOption, describeCharacter, formatCodePoint, describeKey, isBareKeyCharacter, isSpace, skipSpaces, skipSpacesBack, isBlankLine, findNumberEnd, findLineEnd, findBlockStringEnd, abbreviate, LF, SPACE, DOUBLE_QUOTE, HASH, SINGLE_QUOTE, ASTERISK, COMMA, DASH, DOT, SLASH, COLON, OPEN_BRACKET, CLOSE_BRACKET, OPEN_BRACE, CLOSE_BRACE, } from "./shared.js";
|
|
22
4
|
const valueEndCharacters = new Uint8Array(128);
|
|
23
5
|
for (const character of ' \t\n,]}#/') {
|
|
24
6
|
valueEndCharacters[character.codePointAt(0)] = 1;
|
|
25
7
|
}
|
|
26
|
-
const RADIX_DIGIT = {
|
|
27
|
-
x: code => (code >= 0x30 && code <= 0x39) || (code >= 0x41 && code <= 0x46),
|
|
28
|
-
o: code => code >= 0x30 && code <= 0x37,
|
|
29
|
-
b: code => code === 0x30 || code === 0x31,
|
|
30
|
-
};
|
|
31
8
|
/*
|
|
32
|
-
|
|
9
|
+
Whether a UTF-16 code unit may directly follow a scalar value: whitespace, a separator, a closing bracket, a comment, or the end, where `charCodeAt()` returns `NaN`.
|
|
10
|
+
*/
|
|
11
|
+
function isValueEnd(code) {
|
|
12
|
+
return Number.isNaN(code) || (code < 128 && valueEndCharacters[code] === 1);
|
|
13
|
+
}
|
|
14
|
+
/*
|
|
15
|
+
Whether a UTF-16 code unit is a space, a tab, a line feed, or the end.
|
|
16
|
+
*/
|
|
17
|
+
function isSpaceOrLineEnd(code) {
|
|
18
|
+
return isSpace(code) || code === LF || Number.isNaN(code);
|
|
19
|
+
}
|
|
20
|
+
/*
|
|
21
|
+
The lookup tables are maps rather than objects, so that a property added to `Object.prototype` is never read as an entry.
|
|
22
|
+
*/
|
|
23
|
+
const RADIX_DIGIT = new Map([
|
|
24
|
+
['x', code => (code >= 0x30 && code <= 0x39) || (code >= 0x41 && code <= 0x46)],
|
|
25
|
+
['o', code => code >= 0x30 && code <= 0x37],
|
|
26
|
+
['b', code => code === 0x30 || code === 0x31],
|
|
27
|
+
]);
|
|
28
|
+
/*
|
|
29
|
+
The longest token that the instant regular expression and the diagnostic regular expressions look at. Their repeated groups need stack in proportion to the input, so a longer token is rejected with a general message, or described from its first part.
|
|
33
30
|
*/
|
|
34
31
|
const MAX_DIAGNOSED_LENGTH = 1000;
|
|
35
32
|
const RADIX_NAME = { x: 'hexadecimal', o: 'octal', b: 'binary' };
|
|
36
33
|
const RADIX_ARTICLE = { x: 'A', o: 'An', b: 'A' };
|
|
37
34
|
const INSTANT_PREFIX = /^\d{4}-\d{2}-\d{2}/v;
|
|
38
|
-
const INSTANT_START = /^\d{4}-\d{2}-\d{2}T\d{2}:\d/v;
|
|
39
35
|
const DURATION_UNIT_NAMES = DURATION_UNITS.keys().toArray();
|
|
40
36
|
const INSTANT = /^(?<year>\d{4})-(?<month>\d{2})-(?<day>\d{2})T(?<hour>\d{2}):(?<minute>\d{2}):(?<second>\d{2})(?:\.(?<fraction>\d+))?(?<offset>Z|[+\-](?<offsetHour>\d{2}):(?<offsetMinute>\d{2}))$/v;
|
|
37
|
+
const LOCAL_DATE_TIME = /^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d+)?$/v;
|
|
38
|
+
const INSTANT_FORMAT = 'An instant is written as 2026-09-19T14:00:00Z, with an optional fraction of up to nine digits and an offset of Z or ±HH:MM';
|
|
41
39
|
const UNICODE_ESCAPE = /u\{(?<hex>[\da-f]{1,6})\}/vy;
|
|
42
40
|
/*
|
|
43
41
|
Every C0 control character except tab and line feed, and DEL. They are errors anywhere in a document.
|
|
@@ -45,14 +43,23 @@ Every C0 control character except tab and line feed, and DEL. They are errors an
|
|
|
45
43
|
// eslint-disable-next-line no-control-regex, regexp/no-control-character -- Control characters are exactly what must be found.
|
|
46
44
|
const CONTROL_CHARACTER = /[\u{0}-\u{8}\u{B}-\u{1F}\u{7F}]/v;
|
|
47
45
|
const LITERAL_STRING_END = /[\n']/gv;
|
|
48
|
-
const
|
|
46
|
+
const KEY_QUOTING_HINT = '. A key that contains characters other than letters, digits, “_”, and “-” must be quoted';
|
|
47
|
+
/*
|
|
48
|
+
The characters that may be meant as part of a key: visible ones that have no meaning in the grammar, which includes every bare key character. A `=` is left out, because after a key, as in `name=foo`, it was most likely meant as the `:` of INI and TOML, so a key that holds a `=` gets no quoting hint.
|
|
49
|
+
*/
|
|
50
|
+
const QUOTABLE_KEY_CHARACTER = String.raw `[^\p{Default_Ignorable_Code_Point}\p{Other}\p{White_Space}"#'*,.\/:=\[\]\{\}]`;
|
|
51
|
+
const QUOTABLE_KEY_START = new RegExp(`^${QUOTABLE_KEY_CHARACTER}`, 'v');
|
|
52
|
+
/*
|
|
53
|
+
A key made of them, up to its `:`, so the quoting hint can show it quoted.
|
|
54
|
+
*/
|
|
55
|
+
const QUOTABLE_KEY = new RegExp(`^${QUOTABLE_KEY_CHARACTER}+(?=:)`, 'v');
|
|
49
56
|
const ESCAPED_STRING_SPECIAL = /[\n"\\]/gv;
|
|
50
|
-
const SIMPLE_ESCAPES =
|
|
51
|
-
'\\'
|
|
52
|
-
'"'
|
|
53
|
-
n
|
|
54
|
-
t
|
|
55
|
-
|
|
57
|
+
const SIMPLE_ESCAPES = new Map([
|
|
58
|
+
['\\', '\\'],
|
|
59
|
+
['"', '"'],
|
|
60
|
+
['n', '\n'],
|
|
61
|
+
['t', '\t'],
|
|
62
|
+
]);
|
|
56
63
|
/*
|
|
57
64
|
What a backslash followed by one of these characters was probably meant to be.
|
|
58
65
|
*/
|
|
@@ -61,13 +68,34 @@ const ESCAPE_MISTAKES = new Map([
|
|
|
61
68
|
['\'', 'A \' needs no escape inside "..."'],
|
|
62
69
|
['\n', String.raw `A backslash must be followed by an escape character. Use \\ for a literal backslash, or a '...' string`],
|
|
63
70
|
]);
|
|
64
|
-
export function parse(text, options
|
|
65
|
-
const
|
|
66
|
-
validateIntegersOption(integers);
|
|
71
|
+
export function parse(text, options) {
|
|
72
|
+
const integers = getIntegersOption(options);
|
|
67
73
|
const source = decode(text);
|
|
68
74
|
checkCharacters(source);
|
|
75
|
+
// The default `createTime` makes a `Temporal` object of every instant and duration, so no `Time` is left.
|
|
69
76
|
return new Parser(source, integers).parseDocument();
|
|
70
77
|
}
|
|
78
|
+
/*
|
|
79
|
+
An instant, as nanoseconds since the Unix epoch, or a duration, as its length in nanoseconds.
|
|
80
|
+
*/
|
|
81
|
+
export class Time {
|
|
82
|
+
type;
|
|
83
|
+
nanoseconds;
|
|
84
|
+
constructor(type, nanoseconds) {
|
|
85
|
+
this.type = type;
|
|
86
|
+
this.nanoseconds = nanoseconds;
|
|
87
|
+
}
|
|
88
|
+
toTemporal() {
|
|
89
|
+
return this.type === 'Instant' ? new (requireTemporal().Instant)(this.nanoseconds) : createDuration(this.nanoseconds);
|
|
90
|
+
}
|
|
91
|
+
}
|
|
92
|
+
/*
|
|
93
|
+
The same as `parse()` for a string, except that each instant and duration is a `Time` instead of a `Temporal` object, so that it works without `Temporal`. For the tree, which makes the `Temporal` object only when the value is read.
|
|
94
|
+
*/
|
|
95
|
+
export function parseWithTimes(text) {
|
|
96
|
+
checkCharacters(text);
|
|
97
|
+
return new Parser(text, 'bigint', time => time).parseDocument();
|
|
98
|
+
}
|
|
71
99
|
const typedArrayTag = Object.getOwnPropertyDescriptor(Object.getPrototypeOf(Uint8Array.prototype), Symbol.toStringTag).get;
|
|
72
100
|
function decode(text) {
|
|
73
101
|
if (typeof text === 'string') {
|
|
@@ -170,27 +198,19 @@ function defineMember(object, key, value) {
|
|
|
170
198
|
object[key] = value;
|
|
171
199
|
}
|
|
172
200
|
}
|
|
173
|
-
function describeValue(value) {
|
|
174
|
-
if (Array.isArray(value)) {
|
|
175
|
-
return 'an array';
|
|
176
|
-
}
|
|
177
|
-
// Every parsed object is created with `{}`, unlike an instant or a duration.
|
|
178
|
-
return value !== null && typeof value === 'object' && Object.getPrototypeOf(value) === Object.prototype ? 'an object' : 'a value';
|
|
179
|
-
}
|
|
180
|
-
function formatPath(path) {
|
|
181
|
-
return path.map(segment => isBareKey(segment) ? abbreviate(segment) : JSON.stringify(abbreviate(segment))).join('.');
|
|
182
|
-
}
|
|
183
201
|
class Parser {
|
|
184
202
|
#source;
|
|
185
203
|
#index = 0;
|
|
186
204
|
#integers;
|
|
205
|
+
#createTime;
|
|
187
206
|
/*
|
|
188
|
-
|
|
207
|
+
Where the document's collection starts, after the comments and whitespace before it.
|
|
189
208
|
*/
|
|
190
|
-
#
|
|
191
|
-
constructor(source, integers) {
|
|
209
|
+
#documentStart = 0;
|
|
210
|
+
constructor(source, integers, createTime = time => time.toTemporal()) {
|
|
192
211
|
this.#source = source;
|
|
193
212
|
this.#integers = integers;
|
|
213
|
+
this.#createTime = createTime;
|
|
194
214
|
}
|
|
195
215
|
#fail(reason, offset = this.#index) {
|
|
196
216
|
throw ParseError.create(reason, this.#source, offset);
|
|
@@ -218,11 +238,11 @@ class Parser {
|
|
|
218
238
|
}
|
|
219
239
|
let hasLineBreak = this.#skipTrivia();
|
|
220
240
|
while (!this.#isAtEnd()) {
|
|
241
|
+
if (this.#code() === COMMA) {
|
|
242
|
+
this.#fail('Top-level entries are separated by line breaks, not commas. Use braces for a one-line object');
|
|
243
|
+
}
|
|
221
244
|
if (!hasLineBreak) {
|
|
222
|
-
|
|
223
|
-
this.#fail('Top-level entries are separated by line breaks, not commas. Use braces for a one-line object');
|
|
224
|
-
}
|
|
225
|
-
this.#fail(`Expected a line break before the next entry, but found ${this.#describeHere()}`);
|
|
245
|
+
this.#fail(`Expected a line break before the next entry, but found ${this.#describeHere()}${this.#slashCommentHint()}`);
|
|
226
246
|
}
|
|
227
247
|
this.#parseEntry(object, 1);
|
|
228
248
|
hasLineBreak = this.#skipTrivia();
|
|
@@ -230,21 +250,18 @@ class Parser {
|
|
|
230
250
|
return object;
|
|
231
251
|
}
|
|
232
252
|
/*
|
|
233
|
-
A
|
|
253
|
+
A whole document that is one scalar, such as `5` or an instant, fails as an entry. That failure is reported as what it is.
|
|
234
254
|
*/
|
|
235
255
|
#diagnoseBareValue(start) {
|
|
236
256
|
this.#index = start;
|
|
237
257
|
let isBareValue = false;
|
|
238
258
|
try {
|
|
239
259
|
this.#parseValue(1);
|
|
240
|
-
|
|
260
|
+
this.#skipTrivia();
|
|
261
|
+
isBareValue = this.#isAtEnd();
|
|
241
262
|
}
|
|
242
|
-
catch
|
|
243
|
-
//
|
|
244
|
-
if (INSTANT_START.test(this.#source.slice(start, start + 15))) {
|
|
245
|
-
throw error;
|
|
246
|
-
}
|
|
247
|
-
// Otherwise it is not a bare value either, and the original error stands.
|
|
263
|
+
catch {
|
|
264
|
+
// Not a bare value either, so the original error stands.
|
|
248
265
|
}
|
|
249
266
|
if (isBareValue) {
|
|
250
267
|
this.#fail('A bare value is not a document. A document is an object or an array, so write it as `key: value` or `[value]`', start);
|
|
@@ -258,7 +275,7 @@ class Parser {
|
|
|
258
275
|
let hasCrossedLineBreak = false;
|
|
259
276
|
for (;;) {
|
|
260
277
|
const code = source.charCodeAt(this.#index);
|
|
261
|
-
if (code
|
|
278
|
+
if (isSpace(code)) {
|
|
262
279
|
this.#index++;
|
|
263
280
|
}
|
|
264
281
|
else if (code === LF) {
|
|
@@ -296,7 +313,7 @@ class Parser {
|
|
|
296
313
|
const nested = source.indexOf('/*', start + 2);
|
|
297
314
|
// The body may not contain `/*`. An opening that overlaps the closing `*/`, as in `/*/`, is not inside the body.
|
|
298
315
|
if (nested !== -1 && nested + 2 <= end) {
|
|
299
|
-
this.#fail('Block comments cannot be nested, and their body may not contain
|
|
316
|
+
this.#fail('Block comments cannot be nested, and their body may not contain “/*”', nested);
|
|
300
317
|
}
|
|
301
318
|
this.#index = end + 2;
|
|
302
319
|
}
|
|
@@ -305,59 +322,155 @@ class Parser {
|
|
|
305
322
|
*/
|
|
306
323
|
#parseEntry(object, depth) {
|
|
307
324
|
const keyStart = this.#index;
|
|
308
|
-
const
|
|
309
|
-
if (depth + path.length - 1 > MAX_DEPTH) {
|
|
310
|
-
this.#fail(`The document is nested more than ${MAX_DEPTH} levels deep`, keyStart);
|
|
311
|
-
}
|
|
325
|
+
const key = this.#parseKey();
|
|
312
326
|
const code = this.#code();
|
|
327
|
+
if (code === DOT) {
|
|
328
|
+
this.#failDotInKey(keyStart);
|
|
329
|
+
}
|
|
313
330
|
if (code !== COLON) {
|
|
314
331
|
this.#failMissingColon(keyStart);
|
|
315
332
|
}
|
|
333
|
+
const colon = this.#index;
|
|
316
334
|
this.#index++;
|
|
317
335
|
this.#skipTrivia();
|
|
318
|
-
|
|
319
|
-
|
|
336
|
+
let value;
|
|
337
|
+
try {
|
|
338
|
+
value = this.#parseValue(depth + 1);
|
|
339
|
+
}
|
|
340
|
+
catch (error) {
|
|
341
|
+
if (error instanceof ParseError) {
|
|
342
|
+
this.#diagnoseBadValue(error, key, keyStart, colon);
|
|
343
|
+
}
|
|
344
|
+
throw error;
|
|
345
|
+
}
|
|
346
|
+
if (Object.hasOwn(object, key)) {
|
|
347
|
+
this.#fail(`Duplicate key ${describeKey(key)}`, keyStart);
|
|
348
|
+
}
|
|
349
|
+
defineMember(object, key, value);
|
|
320
350
|
}
|
|
321
|
-
|
|
351
|
+
/*
|
|
352
|
+
Reports a value that failed to parse as what the entry was meant to be, when that is clear.
|
|
353
|
+
*/
|
|
354
|
+
#diagnoseBadValue(error, key, keyStart, colon) {
|
|
355
|
+
// A key that contains a `:`, as in `12:30: 'lunch'`, ends at the first `:`, and the rest is read as the value.
|
|
356
|
+
if (isBareKeyCharacter(this.#code(colon + 1))) {
|
|
357
|
+
this.#diagnoseKeyWithColon(keyStart);
|
|
358
|
+
}
|
|
359
|
+
this.#index = colon + 1;
|
|
360
|
+
const isValueOnNextLine = this.#skipTrivia();
|
|
361
|
+
const valueStart = this.#index;
|
|
362
|
+
const commentHint = this.#describeCommentAsValue(colon);
|
|
363
|
+
if (isValueOnNextLine) {
|
|
364
|
+
this.#diagnoseMissingValue(valueStart, key, keyStart, commentHint);
|
|
365
|
+
}
|
|
366
|
+
// A value that was left out at the end of the document or of an object.
|
|
367
|
+
const code = this.#code(valueStart);
|
|
368
|
+
if (commentHint !== '' && error.offset === valueStart && (code === CLOSE_BRACE || Number.isNaN(code))) {
|
|
369
|
+
this.#fail(`${error.reason}${commentHint}`, valueStart);
|
|
370
|
+
}
|
|
371
|
+
}
|
|
372
|
+
/*
|
|
373
|
+
A bare key followed directly by `:` and more of a key, up to a `:` that ends the key, as in `a:b: 1`.
|
|
374
|
+
*/
|
|
375
|
+
#diagnoseKeyWithColon(keyStart) {
|
|
322
376
|
const source = this.#source;
|
|
323
|
-
let
|
|
324
|
-
|
|
325
|
-
|
|
377
|
+
for (let index = keyStart; index < keyStart + MAX_DIAGNOSED_LENGTH; index++) {
|
|
378
|
+
const code = source.charCodeAt(index);
|
|
379
|
+
if (code === COLON) {
|
|
380
|
+
const next = source.charCodeAt(index + 1);
|
|
381
|
+
if (next === LF || isSpace(next) || Number.isNaN(next)) {
|
|
382
|
+
this.#fail(`A key that contains “:” must be quoted, as in '${abbreviate(source.slice(keyStart, index))}'`, keyStart);
|
|
383
|
+
}
|
|
384
|
+
}
|
|
385
|
+
else if (!isBareKeyCharacter(code)) {
|
|
386
|
+
return;
|
|
387
|
+
}
|
|
388
|
+
}
|
|
389
|
+
}
|
|
390
|
+
/*
|
|
391
|
+
The hint for a `#` directly after a `:`, as in `color: #FFF`, which starts a comment rather than a value.
|
|
392
|
+
*/
|
|
393
|
+
#describeCommentAsValue(colon) {
|
|
394
|
+
const source = this.#source;
|
|
395
|
+
const hash = skipSpaces(source, colon + 1);
|
|
396
|
+
if (source.charCodeAt(hash) !== HASH) {
|
|
397
|
+
return '';
|
|
326
398
|
}
|
|
399
|
+
// A `#` followed by a space, or by another `#`, starts an ordinary comment. A separator or a closing bracket after the value is not part of it. The regular expression, which needs stack in proportion to its match, only sees the start of a long comment, which is cut short in the message anyway.
|
|
400
|
+
const match = /^#[^\t\n #,\]\}][^\t\n ,\]\}]*/v.exec(source.slice(hash, hash + MAX_DIAGNOSED_LENGTH));
|
|
401
|
+
return match === null ? '' : `. “#” starts a comment, so a value that starts with “#” must be quoted${quotingExample(match[0])}`;
|
|
402
|
+
}
|
|
403
|
+
/*
|
|
404
|
+
An entry whose value was left out, as in `a:` followed by `'b': 1` on the next line, reads the next key as the value. That failure is reported as what it is.
|
|
405
|
+
*/
|
|
406
|
+
#diagnoseMissingValue(start, parent, keyStart, commentHint) {
|
|
407
|
+
// A YAML block sequence.
|
|
408
|
+
if (this.#code(start) === DASH && isSpace(this.#code(start + 1))) {
|
|
409
|
+
this.#fail('Expected a value, but found a “-” list. An array is written in brackets, as in [80, 443]', start);
|
|
410
|
+
}
|
|
411
|
+
this.#index = start;
|
|
412
|
+
let key;
|
|
413
|
+
try {
|
|
414
|
+
key = this.#parseKey();
|
|
415
|
+
}
|
|
416
|
+
catch {
|
|
417
|
+
// Not a key either, so the original error stands.
|
|
418
|
+
}
|
|
419
|
+
if (key === undefined || this.#code() !== COLON) {
|
|
420
|
+
return;
|
|
421
|
+
}
|
|
422
|
+
// Whitespace or the end follows the `:` of a bare key. A digit follows the `:` in an instant such as `2026-09-19T25:00:00Z`, whose own error is more precise. A quoted key cannot be part of a value, so anything may follow its `:`.
|
|
423
|
+
const next = this.#code(this.#index + 1);
|
|
424
|
+
const firstCode = this.#code(start);
|
|
425
|
+
if (firstCode !== SINGLE_QUOTE && firstCode !== DOUBLE_QUOTE && next !== LF && !isSpace(next) && !Number.isNaN(next)) {
|
|
426
|
+
return;
|
|
427
|
+
}
|
|
428
|
+
const hint = commentHint === '' ? this.#describeIndentedKey(parent, key, keyStart, start) : commentHint;
|
|
429
|
+
this.#fail(`Expected a value, but found the key ${describeKey(key)}${hint}`, start);
|
|
430
|
+
}
|
|
431
|
+
/*
|
|
432
|
+
The hint for a key at `start` that is indented under the entry whose value is missing, as YAML nests an object.
|
|
433
|
+
*/
|
|
434
|
+
#describeIndentedKey(parent, key, keyStart, start) {
|
|
435
|
+
const keyIndentation = getIndentation(this.#source, keyStart);
|
|
436
|
+
const indentation = getIndentation(this.#source, start);
|
|
437
|
+
// The keys are written as in a document, so that the suggestion is valid, rather than as JSON strings like the rest of the message, whose escapes, such as `\b`, are not all valid.
|
|
438
|
+
const hint = `. Indentation does not nest objects, so write ${abbreviate(formatKey(parent), 200)}: {${abbreviate(formatKey(key), 200)}: …}`;
|
|
439
|
+
return keyIndentation === undefined || indentation === undefined || indentation <= keyIndentation ? '' : hint;
|
|
440
|
+
}
|
|
441
|
+
#failMissingColon(keyStart) {
|
|
442
|
+
const source = this.#source;
|
|
443
|
+
const next = skipSpaces(source, this.#index);
|
|
327
444
|
const nextCode = source.charCodeAt(next);
|
|
328
445
|
if (nextCode === COLON && next > this.#index) {
|
|
329
|
-
this.#fail('Whitespace is not allowed between a key and its
|
|
446
|
+
this.#fail('Whitespace is not allowed between a key and its “:”');
|
|
330
447
|
}
|
|
331
|
-
//
|
|
448
|
+
// A block comment followed by the ":" was meant to come before it.
|
|
332
449
|
if (nextCode === SLASH && source.charCodeAt(next + 1) === ASTERISK) {
|
|
333
|
-
|
|
450
|
+
const commentEnd = source.indexOf('*/', next + 2);
|
|
451
|
+
if (commentEnd !== -1 && source.charCodeAt(skipSpaces(source, commentEnd + 2)) === COLON) {
|
|
452
|
+
this.#fail('A comment is not allowed between a key and its “:”', next);
|
|
453
|
+
}
|
|
334
454
|
}
|
|
335
455
|
const colon = source.indexOf(':', next);
|
|
336
|
-
// A key with a space in it, such as `the name: 1`. The hint is only given for plain words, because quoting a
|
|
337
|
-
if (colon !== -1 && next > this.#index && isBareKeyCharacter(nextCode) && colon < findLineEnd(source, next)) {
|
|
338
|
-
|
|
456
|
+
// A key with a space in it, such as `the name: 1`. The hint is only given for plain words, because quoting a quoted key would change what it means, and for a `:` that whitespace or the end follows, because quoting the words before the `:` in `server localhost:8080` would give a valid document with another meaning. So `the name:1` gets no hint.
|
|
457
|
+
if (colon !== -1 && next > this.#index && isBareKeyCharacter(nextCode) && colon < findLineEnd(source, next) && isSpaceOrLineEnd(source.charCodeAt(colon + 1))) {
|
|
458
|
+
// Only spaces and tabs are trimmed, because `trimEnd()` would also remove characters that are not whitespace in SOML, such as U+00A0.
|
|
459
|
+
const key = source.slice(keyStart, skipSpacesBack(source, colon));
|
|
339
460
|
if (isWordsWithSpaces(key)) {
|
|
340
461
|
this.#fail(`A bare key cannot contain spaces. Quote it, as in '${abbreviate(key)}'`, keyStart);
|
|
341
462
|
}
|
|
342
463
|
}
|
|
343
464
|
if (this.#isAtEnd() || this.#code() === LF) {
|
|
344
|
-
this.#fail('Expected
|
|
465
|
+
this.#fail('Expected “:” after the key');
|
|
345
466
|
}
|
|
346
467
|
// A character directly after a bare key is most likely meant to be part of it.
|
|
347
468
|
const previousCode = source.charCodeAt(this.#index - 1);
|
|
348
|
-
const hint = previousCode !== SINGLE_QUOTE && previousCode !== DOUBLE_QUOTE && next === this.#index
|
|
469
|
+
const hint = previousCode !== SINGLE_QUOTE && previousCode !== DOUBLE_QUOTE && next === this.#index && QUOTABLE_KEY_START.test(source.slice(next, next + 2)) ? KEY_QUOTING_HINT : '';
|
|
349
470
|
this.#index = next;
|
|
350
|
-
this.#fail(`Expected
|
|
471
|
+
this.#fail(`Expected “:” after the key, but found ${this.#describeHere()}${hint}`);
|
|
351
472
|
}
|
|
352
473
|
#parseKey() {
|
|
353
|
-
const path = [this.#parseKeySegment()];
|
|
354
|
-
while (this.#code() === DOT) {
|
|
355
|
-
this.#index++;
|
|
356
|
-
path.push(this.#parseKeySegment(true));
|
|
357
|
-
}
|
|
358
|
-
return path;
|
|
359
|
-
}
|
|
360
|
-
#parseKeySegment(isAfterDot = false) {
|
|
361
474
|
const source = this.#source;
|
|
362
475
|
const start = this.#index;
|
|
363
476
|
const code = source.charCodeAt(start);
|
|
@@ -373,100 +486,93 @@ class Parser {
|
|
|
373
486
|
end++;
|
|
374
487
|
}
|
|
375
488
|
if (end === start) {
|
|
376
|
-
if (
|
|
377
|
-
this.#
|
|
489
|
+
if (code === OPEN_BRACKET) {
|
|
490
|
+
this.#diagnoseTableHeader();
|
|
378
491
|
}
|
|
379
|
-
this.#fail(this.#isAtEnd() ? 'Expected a key' : `Expected a key, but found ${this.#describeHere()}`);
|
|
492
|
+
this.#fail(this.#isAtEnd() ? 'Expected a key' : `Expected a key, but found ${this.#describeHere()}${this.#keyQuotingHint()}${this.#slashCommentHint()}`);
|
|
380
493
|
}
|
|
381
494
|
this.#index = end;
|
|
382
495
|
return source.slice(start, end);
|
|
383
496
|
}
|
|
384
|
-
|
|
385
|
-
|
|
386
|
-
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
const child = {};
|
|
391
|
-
this.#dottedObjects.add(child);
|
|
392
|
-
defineMember(target, key, child);
|
|
393
|
-
target = child;
|
|
394
|
-
continue;
|
|
395
|
-
}
|
|
396
|
-
// Only an object built by dotted keys is in the set, which is checked next.
|
|
397
|
-
const existing = target[key];
|
|
398
|
-
if (!this.#dottedObjects.has(existing)) {
|
|
399
|
-
const prefix = formatPath(path.slice(0, index + 1));
|
|
400
|
-
const what = describeValue(existing);
|
|
401
|
-
const reason = what === 'an object' ? 'an object written with braces, which is closed' : what;
|
|
402
|
-
this.#fail(`Cannot set ${formatPath(path)}, because ${prefix} is already ${reason} and a dotted key cannot extend it`, keyStart);
|
|
403
|
-
}
|
|
404
|
-
target = existing;
|
|
497
|
+
/*
|
|
498
|
+
The hint for a key that starts with a character a bare key cannot hold, such as the `$` in `$schema`, or nothing for a character with another meaning, such as `}`.
|
|
499
|
+
*/
|
|
500
|
+
#keyQuotingHint() {
|
|
501
|
+
if (!QUOTABLE_KEY_START.test(this.#source.slice(this.#index, this.#index + 2))) {
|
|
502
|
+
return '';
|
|
405
503
|
}
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
|
|
504
|
+
// A longer key gets the hint without the example, and the regular expression, which backtracks, never sees a huge run.
|
|
505
|
+
const key = QUOTABLE_KEY.exec(this.#source.slice(this.#index, this.#index + MAX_DIAGNOSED_LENGTH))?.[0];
|
|
506
|
+
return key === undefined ? KEY_QUOTING_HINT : `${KEY_QUOTING_HINT}, as in '${abbreviate(key)}'`;
|
|
507
|
+
}
|
|
508
|
+
/*
|
|
509
|
+
A `.` after a key, as in `example.com: 1`. A key is never a path, so the `.` is meant either as part of the key or as nesting.
|
|
510
|
+
*/
|
|
511
|
+
#failDotInKey(keyStart) {
|
|
512
|
+
const source = this.#source;
|
|
513
|
+
let end = this.#index;
|
|
514
|
+
while (end < keyStart + MAX_DIAGNOSED_LENGTH && (isBareKeyCharacter(source.charCodeAt(end)) || source.charCodeAt(end) === DOT)) {
|
|
515
|
+
end++;
|
|
516
|
+
}
|
|
517
|
+
const key = source.slice(keyStart, end);
|
|
518
|
+
const words = key.split('.');
|
|
519
|
+
// The suggestions are only given for a bare key whose dots each have a word on both sides, so that both are valid.
|
|
520
|
+
if (source.charCodeAt(end) !== COLON || !isBareKeyCharacter(source.charCodeAt(keyStart)) || words.includes('')) {
|
|
521
|
+
this.#fail('A key cannot contain “.” unless it is quoted. Quote the whole key, or use braces to nest, as in a: {b: …}');
|
|
412
522
|
}
|
|
413
|
-
|
|
523
|
+
const nested = `${words.join(': {')}: …${'}'.repeat(words.length - 1)}`;
|
|
524
|
+
this.#fail(`A bare key cannot contain “.”. Quote it, as in '${abbreviate(key)}', or use braces to nest, as in ${abbreviate(nested)}`);
|
|
414
525
|
}
|
|
415
526
|
#parseObject(depth) {
|
|
416
|
-
if (depth > MAX_DEPTH) {
|
|
417
|
-
this.#fail(`The document is nested more than ${MAX_DEPTH} levels deep`);
|
|
418
|
-
}
|
|
419
|
-
const start = this.#index;
|
|
420
527
|
const object = {};
|
|
421
|
-
this.#
|
|
422
|
-
this.#skipTrivia();
|
|
423
|
-
for (;;) {
|
|
424
|
-
const code = this.#code();
|
|
425
|
-
if (code === CLOSE_BRACE) {
|
|
426
|
-
this.#index++;
|
|
427
|
-
return object;
|
|
428
|
-
}
|
|
429
|
-
if (this.#isAtEnd()) {
|
|
430
|
-
this.#fail('Unterminated object: expected "}"', start);
|
|
431
|
-
}
|
|
528
|
+
this.#parseItems(depth, CLOSE_BRACE, () => {
|
|
432
529
|
this.#parseEntry(object, depth);
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
this.#index++;
|
|
436
|
-
this.#skipTrivia();
|
|
437
|
-
}
|
|
438
|
-
else if (this.#code() !== CLOSE_BRACE) {
|
|
439
|
-
this.#fail(this.#isAtEnd() ? 'Unterminated object: expected "}"' : `Expected "," or "}" after an object member, but found ${this.#describeHere()}`, this.#isAtEnd() ? start : this.#index);
|
|
440
|
-
}
|
|
441
|
-
}
|
|
530
|
+
});
|
|
531
|
+
return object;
|
|
442
532
|
}
|
|
443
533
|
#parseArray(depth) {
|
|
534
|
+
const array = [];
|
|
535
|
+
this.#parseItems(depth, CLOSE_BRACKET, () => {
|
|
536
|
+
array.push(this.#parseValue(depth + 1));
|
|
537
|
+
});
|
|
538
|
+
return array;
|
|
539
|
+
}
|
|
540
|
+
/*
|
|
541
|
+
`{` or `[` at depth `depth`, then items separated by a comma, a line break, or both, with an optional trailing comma, then the `closing` bracket. A comma must be on the line of the item before it.
|
|
542
|
+
*/
|
|
543
|
+
#parseItems(depth, closing, parseItem) {
|
|
444
544
|
if (depth > MAX_DEPTH) {
|
|
445
545
|
this.#fail(`The document is nested more than ${MAX_DEPTH} levels deep`);
|
|
446
546
|
}
|
|
447
547
|
const start = this.#index;
|
|
448
|
-
const array = [];
|
|
449
548
|
this.#index++;
|
|
450
549
|
this.#skipTrivia();
|
|
451
|
-
|
|
452
|
-
|
|
453
|
-
|
|
454
|
-
|
|
455
|
-
|
|
550
|
+
while (this.#code() !== closing) {
|
|
551
|
+
let hasLineBreak = false;
|
|
552
|
+
let itemEnd = this.#index;
|
|
553
|
+
if (!this.#isAtEnd()) {
|
|
554
|
+
parseItem();
|
|
555
|
+
itemEnd = this.#index;
|
|
556
|
+
hasLineBreak = this.#skipTrivia();
|
|
456
557
|
}
|
|
457
558
|
if (this.#isAtEnd()) {
|
|
458
|
-
this.#fail(
|
|
559
|
+
this.#fail(`Unterminated ${closing === CLOSE_BRACE ? 'object' : 'array'}: expected “${String.fromCharCode(closing)}”`, start);
|
|
459
560
|
}
|
|
460
|
-
array.push(this.#parseValue(depth + 1));
|
|
461
|
-
this.#skipTrivia();
|
|
462
561
|
if (this.#code() === COMMA) {
|
|
562
|
+
if (hasLineBreak) {
|
|
563
|
+
this.#fail('A comma must be on the same line as the item before it. The line break already separates the items, so remove the comma');
|
|
564
|
+
}
|
|
463
565
|
this.#index++;
|
|
464
566
|
this.#skipTrivia();
|
|
465
567
|
}
|
|
466
|
-
else if (this.#code() !==
|
|
467
|
-
|
|
568
|
+
else if (!hasLineBreak && this.#code() !== closing) {
|
|
569
|
+
const item = closing === CLOSE_BRACE ? 'an object member' : 'an array item';
|
|
570
|
+
// Without a line break that separates, a line break in the gap is inside a block comment.
|
|
571
|
+
const hint = this.#source.slice(itemEnd, this.#index).includes('\n') ? '. A line break inside a block comment does not separate items' : this.#slashCommentHint();
|
|
572
|
+
this.#fail(`Expected “,”, a line break, or “${String.fromCharCode(closing)}” after ${item}, but found ${this.#describeHere()}${hint}`);
|
|
468
573
|
}
|
|
469
574
|
}
|
|
575
|
+
this.#index++;
|
|
470
576
|
}
|
|
471
577
|
#parseValue(depth) {
|
|
472
578
|
const code = this.#code();
|
|
@@ -476,7 +582,10 @@ class Parser {
|
|
|
476
582
|
if (code === OPEN_BRACKET) {
|
|
477
583
|
return this.#parseArray(depth);
|
|
478
584
|
}
|
|
585
|
+
const start = this.#index;
|
|
479
586
|
let value;
|
|
587
|
+
// An int or a float written with digits.
|
|
588
|
+
let isNumber = false;
|
|
480
589
|
if (code === SINGLE_QUOTE || code === DOUBLE_QUOTE) {
|
|
481
590
|
value = this.#parseString();
|
|
482
591
|
}
|
|
@@ -497,16 +606,55 @@ class Parser {
|
|
|
497
606
|
}
|
|
498
607
|
else if (code === DASH || isDigit(code)) {
|
|
499
608
|
value = this.#parseNumberOrInstant();
|
|
609
|
+
isNumber = typeof value === 'bigint' || typeof value === 'number';
|
|
500
610
|
}
|
|
501
611
|
else {
|
|
502
612
|
this.#failUnexpectedValue();
|
|
503
613
|
}
|
|
504
|
-
|
|
505
|
-
|
|
506
|
-
|
|
614
|
+
if (!isValueEnd(this.#code())) {
|
|
615
|
+
this.#failAfterValue(start, isNumber);
|
|
616
|
+
}
|
|
617
|
+
if (isNumber && isSpace(this.#code())) {
|
|
618
|
+
this.#diagnoseUnitAfterSpace(start);
|
|
507
619
|
}
|
|
508
620
|
return value;
|
|
509
621
|
}
|
|
622
|
+
/*
|
|
623
|
+
A character that cannot follow the value that starts at `start`.
|
|
624
|
+
*/
|
|
625
|
+
#failAfterValue(start, isNumber) {
|
|
626
|
+
const code = this.#code();
|
|
627
|
+
// Go writes microseconds as `µs`, with the micro sign or the Greek letter mu. The duration is only suggested when it is valid, which a number with an exponent or a radix prefix, for example, is not.
|
|
628
|
+
if (isNumber && (code === 0xB5 || code === 0x3_BC) && this.#code(this.#index + 1) === 0x73 /* s */) {
|
|
629
|
+
const number = this.#source.slice(start, this.#index);
|
|
630
|
+
this.#fail(`The unit for microseconds is written us${isValidValue(`${number}us`) ? `, as in ${abbreviate(number)}us` : ''}`);
|
|
631
|
+
}
|
|
632
|
+
// A `''` inside a '...' string, as SQL and YAML escape a quote, ends the string.
|
|
633
|
+
if (code === SINGLE_QUOTE && this.#code(start) === SINGLE_QUOTE) {
|
|
634
|
+
this.#fail('There is no \'\' escape in a \'...\' string. Write a string that contains \' as "...", as in "it\'s"');
|
|
635
|
+
}
|
|
636
|
+
this.#fail(`Unexpected ${this.#describeHere()} after a value`);
|
|
637
|
+
}
|
|
638
|
+
/*
|
|
639
|
+
A number followed by a space and a word, such as `512 MiB` or `10 seconds`, which is an error anyway. A word followed by more than spaces, a separator, or a comment is left to the general errors, because it may be a key, as in `a: 1 b: 2`, or a sentence.
|
|
640
|
+
*/
|
|
641
|
+
#diagnoseUnitAfterSpace(start) {
|
|
642
|
+
const source = this.#source;
|
|
643
|
+
const unitStart = skipSpaces(source, this.#index);
|
|
644
|
+
let unitEnd = unitStart;
|
|
645
|
+
while (isLetter(source.charCodeAt(unitEnd))) {
|
|
646
|
+
unitEnd++;
|
|
647
|
+
}
|
|
648
|
+
const unit = source.slice(unitStart, unitEnd);
|
|
649
|
+
// A keyword after a number is a missing comma.
|
|
650
|
+
if (unit === '' || !isValueEnd(source.charCodeAt(skipSpaces(source, unitEnd))) || ['true', 'false', 'null', 'infinity'].includes(unit)) {
|
|
651
|
+
return;
|
|
652
|
+
}
|
|
653
|
+
const number = source.slice(start, this.#index);
|
|
654
|
+
// The duration is only suggested when it is valid, which a number with an exponent or a radix prefix, a fraction of a nanosecond, a negative zero, or a value outside the 64-bit range is not. The string is always offered, because a unit such as `m` may mean meters rather than minutes.
|
|
655
|
+
const duration = DURATION_UNITS.has(unit) && isValidValue(`${number}${unit}`) ? `${abbreviate(number)}${unit}` : '10s';
|
|
656
|
+
this.#fail(`A unit cannot follow a number after a space. Write a duration without the space, as in ${duration}, and anything else, such as a size, as a string, as in '${abbreviate(`${number} ${unit}`)}'`, unitStart);
|
|
657
|
+
}
|
|
510
658
|
#isKeyword(word) {
|
|
511
659
|
if (!this.#source.startsWith(word, this.#index)) {
|
|
512
660
|
return false;
|
|
@@ -525,10 +673,11 @@ class Parser {
|
|
|
525
673
|
}
|
|
526
674
|
const code = this.#code();
|
|
527
675
|
if (code === 0x2B /* + */) {
|
|
528
|
-
this.#fail('A
|
|
676
|
+
this.#fail('A “+” sign is not allowed. A number without a sign is positive');
|
|
529
677
|
}
|
|
678
|
+
// A `.` that no digit follows begins a string, such as `.env` or `./foo`, rather than a number.
|
|
530
679
|
if (code === DOT) {
|
|
531
|
-
this.#fail('A number cannot begin with
|
|
680
|
+
this.#fail(isDigit(this.#code(this.#index + 1)) ? 'A number cannot begin with “.”; write a digit before it, as in 0.5' : describeUnknownWord('.', this.#unquotedText()));
|
|
532
681
|
}
|
|
533
682
|
let wordEnd = this.#index;
|
|
534
683
|
while (isBareKeyCharacter(this.#code(wordEnd))) {
|
|
@@ -536,13 +685,66 @@ class Parser {
|
|
|
536
685
|
}
|
|
537
686
|
if (wordEnd > this.#index) {
|
|
538
687
|
const word = this.#source.slice(this.#index, wordEnd);
|
|
539
|
-
//
|
|
540
|
-
|
|
688
|
+
// A key where a value should be, as when the value of an entry is left out and the next line has a key, or as in `[a: 1]`, is reported as a key. A word after the `:` of a member and a space, as in `msg: Error: file not found` or `url: https://example.com`, is that member's value instead, an unquoted string. Without the space, as in `a:b: 1`, the first `:` was most likely meant as part of the key.
|
|
689
|
+
const spacesStart = skipSpacesBack(this.#source, this.#index);
|
|
690
|
+
const isMemberValue = spacesStart < this.#index && this.#code(spacesStart - 1) === COLON;
|
|
691
|
+
if (!isMemberValue && this.#code(wordEnd) === COLON) {
|
|
541
692
|
this.#fail(`Expected a value, but found the key ${abbreviate(word)}`);
|
|
542
693
|
}
|
|
543
|
-
this.#
|
|
694
|
+
this.#diagnoseTableHeader(true);
|
|
695
|
+
this.#fail(describeUnknownWord(word, this.#unquotedText()));
|
|
696
|
+
}
|
|
697
|
+
this.#fail(`Expected a value, but found ${this.#describeHere()}${this.#isBlockScalarIndicator() ? '. Write a multiline string as a block string, between \'\'\' lines' : this.#slashCommentHint()}`);
|
|
698
|
+
}
|
|
699
|
+
/*
|
|
700
|
+
The text from `start` that was most likely meant as one unquoted string, such as `John Smith`: up to the end of the line, a comma, a closing bracket, or a comment after a space or a tab, without the spaces and tabs at its end.
|
|
701
|
+
*/
|
|
702
|
+
#unquotedText(start = this.#index) {
|
|
703
|
+
const source = this.#source;
|
|
704
|
+
const limit = Math.min(source.length, start + MAX_DIAGNOSED_LENGTH);
|
|
705
|
+
let end = start;
|
|
706
|
+
while (end < limit && !isUnquotedTextEnd(source, end)) {
|
|
707
|
+
end++;
|
|
544
708
|
}
|
|
545
|
-
|
|
709
|
+
return source.slice(start, skipSpacesBack(source, end));
|
|
710
|
+
}
|
|
711
|
+
/*
|
|
712
|
+
A TOML table header, as in `[server]` or `[[servers]]` on a line of its own, reads as an array that holds a word, or as a key that starts with "[". Where a value is expected, `isValue`, it is only a table header when its bracket opens the document, because a table cannot be written as an item of an array inside it.
|
|
713
|
+
*/
|
|
714
|
+
#diagnoseTableHeader(isValue = false) {
|
|
715
|
+
const source = this.#source;
|
|
716
|
+
const lineStart = source.lastIndexOf('\n', this.#index - 1) + 1;
|
|
717
|
+
if (isValue && skipSpaces(source, lineStart) !== this.#documentStart) {
|
|
718
|
+
return;
|
|
719
|
+
}
|
|
720
|
+
const line = source.slice(lineStart, Math.min(findLineEnd(source, this.#index), lineStart + MAX_DIAGNOSED_LENGTH));
|
|
721
|
+
// The name is a valid key, so that the suggestion is valid.
|
|
722
|
+
const match = /^[\t ]*\[\[?(?<name>[A-Z_a-z][\w\-]*)\]\]?[\t ]*$/v.exec(line);
|
|
723
|
+
if (match === null) {
|
|
724
|
+
return;
|
|
725
|
+
}
|
|
726
|
+
const { name } = match.groups;
|
|
727
|
+
this.#fail(`There are no table headers. Write the table as an object, as in ${abbreviate(name)}: {…}`);
|
|
728
|
+
}
|
|
729
|
+
/*
|
|
730
|
+
A YAML literal block scalar indicator, as in `key: |` or `key: |-`, at the end of its line.
|
|
731
|
+
*/
|
|
732
|
+
#isBlockScalarIndicator() {
|
|
733
|
+
const code = this.#code();
|
|
734
|
+
// A folded block scalar, `>`, joins its lines, which a block string does not, so only `|` gets the hint.
|
|
735
|
+
if (code !== 0x7C /* | */) {
|
|
736
|
+
return false;
|
|
737
|
+
}
|
|
738
|
+
const chomping = this.#code(this.#index + 1);
|
|
739
|
+
const end = skipSpaces(this.#source, chomping === DASH || chomping === 0x2B /* + */ ? this.#index + 2 : this.#index + 1);
|
|
740
|
+
const next = this.#code(end);
|
|
741
|
+
return next === LF || Number.isNaN(next);
|
|
742
|
+
}
|
|
743
|
+
/*
|
|
744
|
+
The hint for a `//` comment, as in JavaScript.
|
|
745
|
+
*/
|
|
746
|
+
#slashCommentHint() {
|
|
747
|
+
return this.#code() === SLASH && this.#code(this.#index + 1) === SLASH ? '. A comment starts with “#”' : '';
|
|
546
748
|
}
|
|
547
749
|
#describeHere() {
|
|
548
750
|
if (this.#isAtEnd()) {
|
|
@@ -602,8 +804,9 @@ class Parser {
|
|
|
602
804
|
#parseEscape(offset) {
|
|
603
805
|
const source = this.#source;
|
|
604
806
|
const character = source[offset + 1];
|
|
605
|
-
|
|
606
|
-
|
|
807
|
+
const simple = SIMPLE_ESCAPES.get(character ?? '');
|
|
808
|
+
if (simple !== undefined) {
|
|
809
|
+
return { text: simple, end: offset + 2 };
|
|
607
810
|
}
|
|
608
811
|
if (character === 'u') {
|
|
609
812
|
UNICODE_ESCAPE.lastIndex = offset + 1;
|
|
@@ -612,12 +815,10 @@ class Parser {
|
|
|
612
815
|
this.#fail(describeBadUnicodeEscape(source, offset + 1), offset);
|
|
613
816
|
}
|
|
614
817
|
const { hex } = match.groups;
|
|
615
|
-
if (hex.length > 1 && hex.startsWith('0')) {
|
|
616
|
-
this.#fail(String.raw `A Unicode escape may not have leading zeros; write \u{${hex.replace(/^0+(?=.)/v, '')}}`, offset);
|
|
617
|
-
}
|
|
618
818
|
const codePoint = Number.parseInt(hex, 16);
|
|
819
|
+
// The value is checked before the spelling, so that the leading zeros error never suggests an escape that is not allowed either.
|
|
619
820
|
if (codePoint === 0x0D) {
|
|
620
|
-
this.#fail(String.raw `A carriage return (U+000D) cannot be represented, so \u{
|
|
821
|
+
this.#fail(String.raw `A carriage return (U+000D) cannot be represented, so \u{${hex}} is not allowed`, offset);
|
|
621
822
|
}
|
|
622
823
|
if (codePoint >= 0xD8_00 && codePoint <= 0xDF_FF) {
|
|
623
824
|
this.#fail(String.raw `\u{${hex}} is a surrogate, which is not a Unicode scalar value`, offset);
|
|
@@ -625,10 +826,13 @@ class Parser {
|
|
|
625
826
|
if (codePoint > 0x10_FF_FF) {
|
|
626
827
|
this.#fail(String.raw `\u{${hex}} is above U+10FFFF, the largest Unicode scalar value`, offset);
|
|
627
828
|
}
|
|
829
|
+
if (hex.length > 1 && hex.startsWith('0')) {
|
|
830
|
+
this.#fail(String.raw `A Unicode escape may not have leading zeros; write \u{${hex.replace(/^0+(?=.)/v, '')}}`, offset);
|
|
831
|
+
}
|
|
628
832
|
return { text: String.fromCodePoint(codePoint), end: offset + 1 + match[0].length };
|
|
629
833
|
}
|
|
630
834
|
// A backslash at the end of the document is the same mistake as one at the end of a line.
|
|
631
|
-
const reason = ESCAPE_MISTAKES.get(character ?? '\n') ?? `Unknown escape
|
|
835
|
+
const reason = ESCAPE_MISTAKES.get(character ?? '\n') ?? `Unknown escape “\\${String.fromCodePoint(source.codePointAt(offset + 1))}”. The escapes are \\\\, \\", \\n, \\t, and \\u{…}; use a '...' string for literal backslashes`;
|
|
632
836
|
this.#fail(reason, offset);
|
|
633
837
|
}
|
|
634
838
|
/*
|
|
@@ -652,7 +856,7 @@ class Parser {
|
|
|
652
856
|
const contentStart = this.#index + 1;
|
|
653
857
|
const closing = findBlockStringEnd(source, contentStart, quote, delimiterLength);
|
|
654
858
|
if (closing === undefined) {
|
|
655
|
-
this.#fail(
|
|
859
|
+
this.#fail(`Unterminated block string${this.#describeInlineClosingDelimiter(contentStart, quote, delimiterLength)}`, start);
|
|
656
860
|
}
|
|
657
861
|
const indentation = source.slice(closing.lineStart, closing.delimiterStart);
|
|
658
862
|
this.#index = closing.delimiterStart + delimiterLength;
|
|
@@ -660,7 +864,7 @@ class Parser {
|
|
|
660
864
|
for (let lineStart = contentStart; lineStart < closing.lineStart; lineStart = findLineEnd(source, lineStart) + 1) {
|
|
661
865
|
const line = source.slice(lineStart, findLineEnd(source, lineStart));
|
|
662
866
|
// A blank line, which is empty or holds only spaces and tabs, may leave out the indentation, and it becomes an empty line. Swift is stricter here: there only a completely empty line may leave it out.
|
|
663
|
-
if (
|
|
867
|
+
if (isBlankLine(line)) {
|
|
664
868
|
contents.push({ text: '', offset: lineStart });
|
|
665
869
|
}
|
|
666
870
|
else if (line.startsWith(indentation)) {
|
|
@@ -682,6 +886,24 @@ class Parser {
|
|
|
682
886
|
const kept = contents.slice(first, last);
|
|
683
887
|
return quote === SINGLE_QUOTE ? kept.map(line => line.text).join('\n') : kept.map(line => this.#unescapeLine(line.text, line.offset)).join('\n');
|
|
684
888
|
}
|
|
889
|
+
/*
|
|
890
|
+
The hint for a content line that ends with the delimiter, as TOML allows. Adding a closing line after it would keep the delimiter as content.
|
|
891
|
+
*/
|
|
892
|
+
#describeInlineClosingDelimiter(contentStart, quote, delimiterLength) {
|
|
893
|
+
const source = this.#source;
|
|
894
|
+
const quoteCharacter = String.fromCharCode(quote);
|
|
895
|
+
const delimiter = quoteCharacter.repeat(delimiterLength);
|
|
896
|
+
let line = source.slice(0, contentStart).split('\n').length;
|
|
897
|
+
for (let lineStart = contentStart; lineStart < source.length; lineStart = findLineEnd(source, lineStart) + 1) {
|
|
898
|
+
const text = source.slice(skipSpaces(source, lineStart), skipSpacesBack(source, findLineEnd(source, lineStart)));
|
|
899
|
+
// A line of only quotes, longer than the delimiter, is content.
|
|
900
|
+
if (text.endsWith(delimiter) && text !== quoteCharacter.repeat(text.length)) {
|
|
901
|
+
return `. Its closing delimiter must start a line, so move the ${delimiter} at the end of line ${line} to a new line`;
|
|
902
|
+
}
|
|
903
|
+
line++;
|
|
904
|
+
}
|
|
905
|
+
return '';
|
|
906
|
+
}
|
|
685
907
|
#unescapeLine(text, offset) {
|
|
686
908
|
let value = '';
|
|
687
909
|
let chunkStart = 0;
|
|
@@ -710,17 +932,16 @@ class Parser {
|
|
|
710
932
|
return this.#parseDuration(text, start);
|
|
711
933
|
}
|
|
712
934
|
switch (kind) {
|
|
713
|
-
case '
|
|
714
|
-
case 'radix': {
|
|
935
|
+
case 'int': {
|
|
715
936
|
if (text === '-0') {
|
|
716
|
-
this.#fail('
|
|
937
|
+
this.#fail('“-0” is not allowed, because zero has one spelling: 0', start);
|
|
717
938
|
}
|
|
718
939
|
return this.#integer(text.replaceAll('_', ''), text, start);
|
|
719
940
|
}
|
|
720
941
|
case 'float': {
|
|
721
942
|
const value = Number(text.replaceAll('_', ''));
|
|
722
943
|
if (!Number.isFinite(value)) {
|
|
723
|
-
this.#fail(`${abbreviate(text)} is too large to be a finite float. Use infinity if you mean it`, start);
|
|
944
|
+
this.#fail(`${abbreviate(text)} is too large to be a finite float. Use ${value < 0 ? '-infinity' : 'infinity'} if you mean it`, start);
|
|
724
945
|
}
|
|
725
946
|
if (value === 0) {
|
|
726
947
|
// A nonzero digit before the exponent means the literal is not zero, so it underflowed.
|
|
@@ -736,7 +957,49 @@ class Parser {
|
|
|
736
957
|
break;
|
|
737
958
|
}
|
|
738
959
|
}
|
|
739
|
-
this.#fail(describeBadNumber(text), start);
|
|
960
|
+
this.#fail(this.#describeThousandsSeparator(text, start) ?? describeBadNumber(text, this.#unquotedText(start)), start);
|
|
961
|
+
}
|
|
962
|
+
/*
|
|
963
|
+
A comma used as a thousands separator, as in `[1,000]`, makes the group after it a separate item, which fails only when it has a leading zero. Writing that item in octal, as the leading zero message suggests, would parse to the wrong items.
|
|
964
|
+
*/
|
|
965
|
+
#describeThousandsSeparator(text, start) {
|
|
966
|
+
if (!/^0\d{2}$/v.test(text) || this.#code(start - 1) !== COMMA) {
|
|
967
|
+
return undefined;
|
|
968
|
+
}
|
|
969
|
+
const getGroupStart = (groupEnd) => {
|
|
970
|
+
let groupStart = groupEnd;
|
|
971
|
+
while (isDigit(this.#code(groupStart - 1))) {
|
|
972
|
+
groupStart--;
|
|
973
|
+
}
|
|
974
|
+
return groupStart;
|
|
975
|
+
};
|
|
976
|
+
let groupEnd = start - 1;
|
|
977
|
+
let numberStart = getGroupStart(groupEnd);
|
|
978
|
+
// Earlier groups of three digits, as in `12,345,000`.
|
|
979
|
+
while (groupEnd - numberStart === 3 && this.#code(numberStart - 1) === COMMA && isDigit(this.#code(numberStart - 2))) {
|
|
980
|
+
groupEnd = numberStart - 1;
|
|
981
|
+
numberStart = getGroupStart(groupEnd);
|
|
982
|
+
}
|
|
983
|
+
const before = this.#code(numberStart - 1);
|
|
984
|
+
// The first group has one to three digits, which are not the end of a float, of a number with underscores, or of a number with a radix prefix.
|
|
985
|
+
if (numberStart === groupEnd
|
|
986
|
+
|| before === DOT
|
|
987
|
+
|| before === 0x5F /* _ */
|
|
988
|
+
|| numberStart < groupEnd - 3
|
|
989
|
+
|| isLetter(before)) {
|
|
990
|
+
return undefined;
|
|
991
|
+
}
|
|
992
|
+
// The sign belongs to the number, as in `-1,000`.
|
|
993
|
+
if (before === DASH) {
|
|
994
|
+
numberStart--;
|
|
995
|
+
}
|
|
996
|
+
const groupsStart = start + text.length;
|
|
997
|
+
const groups = /^(?:,\d{3}(?!\d))*/v.exec(this.#source.slice(groupsStart, groupsStart + MAX_DIAGNOSED_LENGTH))[0];
|
|
998
|
+
const number = this.#source.slice(numberStart, groupsStart + groups.length);
|
|
999
|
+
const reason = `Leading zeros are not allowed in a decimal number. A comma separates items, so ${abbreviate(number)} is not one number`;
|
|
1000
|
+
// Both spellings are the same int, so they are suggested only when it is in range.
|
|
1001
|
+
const withUnderscores = number.replaceAll(',', '_');
|
|
1002
|
+
return isValidValue(withUnderscores) ? `${reason}. Write it as ${abbreviate(withUnderscores)} or ${abbreviate(number.replaceAll(',', ''))}` : reason;
|
|
740
1003
|
}
|
|
741
1004
|
/*
|
|
742
1005
|
The common case, a short decimal int or float such as `8080`, `-3`, or `30.5`, without the general path's maximal-run scan and classification. Anything else, including every error, returns `undefined` and is left to the general path.
|
|
@@ -750,7 +1013,7 @@ class Parser {
|
|
|
750
1013
|
index++;
|
|
751
1014
|
}
|
|
752
1015
|
const integerLength = index - integerStart;
|
|
753
|
-
// No digits, or a leading zero, which
|
|
1016
|
+
// No digits, or a leading zero followed by more digits, which the general path reports as an error or reads as an instant before the year 1000.
|
|
754
1017
|
if (integerLength === 0 || (integerLength > 1 && source.charCodeAt(integerStart) === 0x30)) {
|
|
755
1018
|
return;
|
|
756
1019
|
}
|
|
@@ -766,9 +1029,8 @@ class Parser {
|
|
|
766
1029
|
}
|
|
767
1030
|
isFloat = true;
|
|
768
1031
|
}
|
|
769
|
-
const next = source.charCodeAt(index);
|
|
770
1032
|
// At most 15 digits, so an int is exact as a number. A character that may continue a number, such as `e`, `_`, or `-`, needs the general path.
|
|
771
|
-
if (index - integerStart > 15 || (
|
|
1033
|
+
if (index - integerStart > 15 || !isValueEnd(source.charCodeAt(index))) {
|
|
772
1034
|
return;
|
|
773
1035
|
}
|
|
774
1036
|
const value = Number(source.slice(start, index));
|
|
@@ -811,7 +1073,14 @@ class Parser {
|
|
|
811
1073
|
#parseInstant(text, start) {
|
|
812
1074
|
const match = text.length > MAX_DIAGNOSED_LENGTH ? null : INSTANT.exec(text);
|
|
813
1075
|
if (match === null) {
|
|
814
|
-
|
|
1076
|
+
// A date, a space, and a time, as TOML and Python write a date and time.
|
|
1077
|
+
const timeStart = this.#index + 1;
|
|
1078
|
+
const end = this.#code() === SPACE && isDigit(this.#code(timeStart)) ? findNumberEnd(this.#source, timeStart) : this.#index;
|
|
1079
|
+
const time = this.#source.slice(timeStart, end);
|
|
1080
|
+
// More of the instant may follow, as in `14:00:00,5Z` or `14:00:00[Europe/Oslo]`, so it may have an offset.
|
|
1081
|
+
const next = this.#code(end);
|
|
1082
|
+
const isWholeValue = isValueEnd(next) && !(next === COMMA && isDigit(this.#code(end + 1)));
|
|
1083
|
+
this.#fail(describeBadInstant(text, time, isWholeValue), start);
|
|
815
1084
|
}
|
|
816
1085
|
const { year, month, day, hour, minute, second, fraction, offset, offsetHour, offsetMinute } = match.groups;
|
|
817
1086
|
const check = (isValid, reason) => {
|
|
@@ -830,12 +1099,13 @@ class Parser {
|
|
|
830
1099
|
if (offset !== 'Z') {
|
|
831
1100
|
check(offsetHour <= '23', 'the offset hour must be 00 to 23');
|
|
832
1101
|
check(offsetMinute <= '59', 'the offset minute must be 00 to 59');
|
|
833
|
-
check(offset !== '-00:00', '-00:00 means
|
|
1102
|
+
check(offset !== '-00:00', '-00:00 means “offset unknown” in RFC 3339, which is not representable; use Z or +00:00');
|
|
834
1103
|
}
|
|
835
|
-
|
|
836
|
-
const
|
|
1104
|
+
// `Date.parse()` reads this format exactly, for every year from 0001 to 9999 and every offset, so the range check needs no `Temporal`.
|
|
1105
|
+
const milliseconds = Date.parse(`${year}-${month}-${day}T${hour}:${minute}:${second}${offset}`);
|
|
1106
|
+
const nanoseconds = (BigInt(milliseconds) * 1000000n) + BigInt((fraction ?? '').padEnd(9, '0'));
|
|
837
1107
|
check(nanoseconds >= MIN_INSTANT && nanoseconds <= MAX_INSTANT, 'in UTC it falls outside the years 0001 to 9999');
|
|
838
|
-
return
|
|
1108
|
+
return this.#createTime(new Time('Instant', nanoseconds));
|
|
839
1109
|
}
|
|
840
1110
|
/*
|
|
841
1111
|
A scan of the parts in order, so that each error names the part it is about.
|
|
@@ -859,7 +1129,7 @@ class Parser {
|
|
|
859
1129
|
if (code === DASH && isDigit(text.charCodeAt(index + 1))) {
|
|
860
1130
|
fail('only the whole duration takes a sign, as in -1h30m');
|
|
861
1131
|
}
|
|
862
|
-
fail(code === 0x2B /* + */ ? 'a
|
|
1132
|
+
fail(code === 0x2B /* + */ ? 'a “+” sign is not allowed' : `expected a number at “${abbreviate(text.slice(index), 10)}”`);
|
|
863
1133
|
}
|
|
864
1134
|
if (text.charCodeAt(index) === 0x30 && (next === 0x5F /* _ */ || isDigit(next))) {
|
|
865
1135
|
fail('leading zeros are not allowed');
|
|
@@ -872,7 +1142,7 @@ class Parser {
|
|
|
872
1142
|
if (next === DOT) {
|
|
873
1143
|
unitStart = skipDigits(text, digitsEnd + 1, isDigit);
|
|
874
1144
|
if (unitStart === digitsEnd + 1) {
|
|
875
|
-
fail('a
|
|
1145
|
+
fail('a “.” must be followed by a digit');
|
|
876
1146
|
}
|
|
877
1147
|
if (text.charCodeAt(unitStart) === 0x5F /* _ */) {
|
|
878
1148
|
fail('an underscore must be between two digits');
|
|
@@ -887,7 +1157,7 @@ class Parser {
|
|
|
887
1157
|
const unit = text.slice(unitStart, unitEnd);
|
|
888
1158
|
const rank = DURATION_UNIT_NAMES.indexOf(unit);
|
|
889
1159
|
if (rank === -1) {
|
|
890
|
-
fail(describeBadDurationUnit(unit));
|
|
1160
|
+
fail(describeBadDurationUnit(unit, text));
|
|
891
1161
|
}
|
|
892
1162
|
if (rank <= previousRank) {
|
|
893
1163
|
fail('the units are in the order h, m, s, ms, us, ns, and each appears at most once');
|
|
@@ -915,15 +1185,16 @@ class Parser {
|
|
|
915
1185
|
total += fractionNanoseconds / scale;
|
|
916
1186
|
}
|
|
917
1187
|
if (isNegative && total === 0n) {
|
|
918
|
-
fail('
|
|
1188
|
+
fail('“-” is not allowed before zero, because zero has one spelling: 0s');
|
|
919
1189
|
}
|
|
920
1190
|
if (total > (isNegative ? -INT64_MIN : INT64_MAX)) {
|
|
921
1191
|
fail('it is outside the 64-bit range of nanoseconds, about 292 years either way');
|
|
922
1192
|
}
|
|
923
|
-
return
|
|
1193
|
+
return this.#createTime(new Time('Duration', isNegative ? -total : total));
|
|
924
1194
|
}
|
|
925
1195
|
parseDocument() {
|
|
926
1196
|
this.#skipTrivia();
|
|
1197
|
+
this.#documentStart = this.#index;
|
|
927
1198
|
if (this.#isAtEnd()) {
|
|
928
1199
|
this.#fail('A document must contain an object or an array, but this one is empty');
|
|
929
1200
|
}
|
|
@@ -946,13 +1217,24 @@ class Parser {
|
|
|
946
1217
|
}
|
|
947
1218
|
}
|
|
948
1219
|
/*
|
|
949
|
-
Whether `text` is
|
|
1220
|
+
Whether `text` is one valid value, so that an error message only suggests a fix that works.
|
|
1221
|
+
*/
|
|
1222
|
+
function isValidValue(text) {
|
|
1223
|
+
try {
|
|
1224
|
+
return new Parser(`[${text}]`, 'bigint', time => time).parseDocument().length === 1;
|
|
1225
|
+
}
|
|
1226
|
+
catch {
|
|
1227
|
+
return false;
|
|
1228
|
+
}
|
|
1229
|
+
}
|
|
1230
|
+
/*
|
|
1231
|
+
Whether `text` is an int, in any radix, or a float, following the grammar. A run of digits may have single underscores between digits.
|
|
950
1232
|
*/
|
|
951
1233
|
function classifyNumber(text) {
|
|
952
|
-
const radix = text.charCodeAt(0) === 0x30 ? RADIX_DIGIT
|
|
1234
|
+
const radix = text.charCodeAt(0) === 0x30 ? RADIX_DIGIT.get(text.charAt(1)) : undefined;
|
|
953
1235
|
if (radix !== undefined) {
|
|
954
1236
|
const end = skipDigits(text, 2, radix);
|
|
955
|
-
return end > 2 && end === text.length ? '
|
|
1237
|
+
return end > 2 && end === text.length ? 'int' : undefined;
|
|
956
1238
|
}
|
|
957
1239
|
let index = text.charCodeAt(0) === DASH ? 1 : 0;
|
|
958
1240
|
// The integer part is `0` or has no leading zero.
|
|
@@ -961,7 +1243,7 @@ function classifyNumber(text) {
|
|
|
961
1243
|
return undefined;
|
|
962
1244
|
}
|
|
963
1245
|
index = integerEnd;
|
|
964
|
-
let kind = '
|
|
1246
|
+
let kind = 'int';
|
|
965
1247
|
if (text.charCodeAt(index) === DOT) {
|
|
966
1248
|
const end = skipDigits(text, index + 1, isDigit);
|
|
967
1249
|
if (end === index + 1) {
|
|
@@ -1037,7 +1319,7 @@ function isDurationLike(text) {
|
|
|
1037
1319
|
}
|
|
1038
1320
|
return /^[dhmnsuw]$/iv.test(text.charAt(index));
|
|
1039
1321
|
}
|
|
1040
|
-
function describeBadDurationUnit(unit) {
|
|
1322
|
+
function describeBadDurationUnit(unit, text) {
|
|
1041
1323
|
if (unit === '') {
|
|
1042
1324
|
return 'every number needs a unit: h, m, s, ms, us, or ns';
|
|
1043
1325
|
}
|
|
@@ -1047,12 +1329,24 @@ function describeBadDurationUnit(unit) {
|
|
|
1047
1329
|
if (/^w(?:eeks?)?$/v.test(unit)) {
|
|
1048
1330
|
return 'there is no week unit, because a day is not a fixed length. Write 168h for a fixed 168 hours';
|
|
1049
1331
|
}
|
|
1050
|
-
|
|
1332
|
+
if (unit === 'M') {
|
|
1333
|
+
// A number with `M` alone is more often a size, as in `memory: 512M`, and `512m` would read as minutes.
|
|
1334
|
+
const sizeHint = text.length <= MAX_DIAGNOSED_LENGTH && /^\d[\d_]*M$/v.test(text) ? `. A size, such as 512M, is a string: '${abbreviate(text)}'` : '';
|
|
1335
|
+
return `there is no month unit, because a month is not a fixed length${sizeHint}`;
|
|
1336
|
+
}
|
|
1337
|
+
return DURATION_UNITS.has(unit.toLowerCase()) ? `the units are lowercase: ${unit.toLowerCase()}` : `“${abbreviate(unit, 10)}” is not a unit. The units are h, m, s, ms, us, and ns, and a string must be quoted`;
|
|
1338
|
+
}
|
|
1339
|
+
/*
|
|
1340
|
+
The number of spaces and tabs before `offset` on its line, or `undefined` when something else comes before it.
|
|
1341
|
+
*/
|
|
1342
|
+
function getIndentation(source, offset) {
|
|
1343
|
+
const lineStart = source.lastIndexOf('\n', offset - 1) + 1;
|
|
1344
|
+
return skipSpaces(source, lineStart) === offset ? offset - lineStart : undefined;
|
|
1051
1345
|
}
|
|
1052
1346
|
function isWordsWithSpaces(text) {
|
|
1053
1347
|
for (let index = 0; index < text.length; index++) {
|
|
1054
1348
|
const code = text.charCodeAt(index);
|
|
1055
|
-
if (code
|
|
1349
|
+
if (!isSpace(code) && !isBareKeyCharacter(code)) {
|
|
1056
1350
|
return false;
|
|
1057
1351
|
}
|
|
1058
1352
|
}
|
|
@@ -1065,23 +1359,38 @@ function daysInMonth(year, month) {
|
|
|
1065
1359
|
}
|
|
1066
1360
|
return [4, 6, 9, 11].includes(month) ? 30 : 31;
|
|
1067
1361
|
}
|
|
1068
|
-
function
|
|
1069
|
-
const
|
|
1362
|
+
function isUnquotedTextEnd(source, index) {
|
|
1363
|
+
const code = source.charCodeAt(index);
|
|
1364
|
+
const isComment = code === HASH || (code === SLASH && source.charCodeAt(index + 1) === ASTERISK);
|
|
1365
|
+
return code === LF || code === COMMA || code === CLOSE_BRACKET || code === CLOSE_BRACE || (isComment && isSpace(source.charCodeAt(index - 1)));
|
|
1366
|
+
}
|
|
1367
|
+
/*
|
|
1368
|
+
The example in a message that a string must be quoted, which quotes `text` as a `'...'` string. That cannot hold a `'`, so then there is no simple example.
|
|
1369
|
+
*/
|
|
1370
|
+
function quotingExample(text) {
|
|
1371
|
+
return text.includes('\'') ? '' : `, as in '${abbreviate(text)}'`;
|
|
1372
|
+
}
|
|
1373
|
+
/*
|
|
1374
|
+
`text` is the unquoted string that `word` begins, which the suggestion quotes.
|
|
1375
|
+
*/
|
|
1376
|
+
function describeUnknownWord(word, text) {
|
|
1377
|
+
// A keyword hint is only for the word on its own. In `Yes please` or `Nan Goldin`, following it would leave text behind or change the value.
|
|
1378
|
+
const lowercase = text === word ? word.toLowerCase() : '';
|
|
1070
1379
|
if (['true', 'false', 'yes', 'no', 'on', 'off'].includes(lowercase)) {
|
|
1071
|
-
return
|
|
1380
|
+
return `“${word}” is not a value. Booleans are written true and false, in lowercase`;
|
|
1072
1381
|
}
|
|
1073
1382
|
if (['null', 'nil', 'none', 'undefined'].includes(lowercase)) {
|
|
1074
|
-
return
|
|
1383
|
+
return `“${word}” is not a value. Null is written null, in lowercase`;
|
|
1075
1384
|
}
|
|
1076
1385
|
if (lowercase === 'nan') {
|
|
1077
1386
|
return 'NaN is not representable. Use null for a missing value';
|
|
1078
1387
|
}
|
|
1079
|
-
return ['inf', 'infinity'].includes(lowercase) ?
|
|
1388
|
+
return ['inf', 'infinity'].includes(lowercase) ? `“${word}” is not a value. Infinity is written infinity, in lowercase` : `Unexpected “${abbreviate(word)}”. A string value must be quoted${quotingExample(text)}`;
|
|
1080
1389
|
}
|
|
1081
1390
|
function describeBadUnicodeEscape(source, offset) {
|
|
1082
1391
|
const rest = source.slice(offset, offset + 12);
|
|
1083
1392
|
if (/^u[\dA-Fa-f]{4}/v.test(rest)) {
|
|
1084
|
-
return
|
|
1393
|
+
return describeFourDigitEscape(rest);
|
|
1085
1394
|
}
|
|
1086
1395
|
if (/^u\{[\dA-Fa-f]*[A-F]/v.test(rest)) {
|
|
1087
1396
|
return 'A Unicode escape uses lowercase hexadecimal digits';
|
|
@@ -1091,15 +1400,36 @@ function describeBadUnicodeEscape(source, offset) {
|
|
|
1091
1400
|
}
|
|
1092
1401
|
return /^u\{[\da-f]{7}/v.test(rest) ? 'A Unicode escape has at most six hexadecimal digits' : String.raw `A Unicode escape is written \u{…} with one to six lowercase hexadecimal digits`;
|
|
1093
1402
|
}
|
|
1094
|
-
|
|
1403
|
+
/*
|
|
1404
|
+
The JSON form `\uXXXX`, from its `u`, with the escape to write instead. JSON writes a character above U+FFFF as two of them, a surrogate pair, which is one `\u{…}` escape here. Here, a lone surrogate and a carriage return have no escape.
|
|
1405
|
+
*/
|
|
1406
|
+
function describeFourDigitEscape(rest) {
|
|
1407
|
+
const code = Number.parseInt(rest.slice(1, 5), 16);
|
|
1408
|
+
const low = /^\\u[\dA-Fa-f]{4}/v.test(rest.slice(5)) ? rest.slice(7, 11) : undefined;
|
|
1409
|
+
const lowCode = low === undefined ? NaN : Number.parseInt(low, 16);
|
|
1410
|
+
if (code >= 0xD8_00 && code <= 0xDB_FF && lowCode >= 0xDC_00 && lowCode <= 0xDF_FF) {
|
|
1411
|
+
const codePoint = 0x1_00_00 + ((code - 0xD8_00) * 0x4_00) + (lowCode - 0xDC_00);
|
|
1412
|
+
return `The four-digit \\${rest.slice(0, 5)}\\u${low} form is not an escape. Write \\u{${codePoint.toString(16)}}`;
|
|
1413
|
+
}
|
|
1414
|
+
const form = `The four-digit \\${rest.slice(0, 5)} form is not an escape`;
|
|
1415
|
+
if (code >= 0xD8_00 && code <= 0xDF_FF) {
|
|
1416
|
+
return String.raw `${form}, and a lone surrogate is not a Unicode scalar value. Write the character it is half of as one \u{…} escape`;
|
|
1417
|
+
}
|
|
1418
|
+
return code === 0x0D ? `${form}, and a carriage return (U+000D) cannot be represented` : String.raw `${form}. Write \u{${code.toString(16)}}`;
|
|
1419
|
+
}
|
|
1420
|
+
/*
|
|
1421
|
+
`unquotedText` is the unquoted string that `fullText` begins, for the suggestion to quote it.
|
|
1422
|
+
*/
|
|
1423
|
+
function describeBadNumber(fullText, unquotedText) {
|
|
1095
1424
|
const text = fullText.slice(0, MAX_DIAGNOSED_LENGTH);
|
|
1096
1425
|
if (text.includes('+')) {
|
|
1097
|
-
return 'A
|
|
1426
|
+
return 'A “+” sign is not allowed in a number, including in an exponent';
|
|
1098
1427
|
}
|
|
1099
1428
|
if (/^-?0[BOX]/v.test(text)) {
|
|
1100
1429
|
return 'A number prefix is lowercase: 0x, 0o, or 0b';
|
|
1101
1430
|
}
|
|
1102
|
-
|
|
1431
|
+
// The whole number, not the part cut at the diagnosed length, because a cut can end before the digit or the underscore that the message is about. Each check below takes linear time.
|
|
1432
|
+
const radixMatch = /^(?<sign>-?)0(?<radix>[box])(?<digits>.*)$/v.exec(fullText);
|
|
1103
1433
|
if (radixMatch !== null) {
|
|
1104
1434
|
const { sign, radix, digits } = radixMatch.groups;
|
|
1105
1435
|
const name = RADIX_NAME[radix];
|
|
@@ -1107,26 +1437,44 @@ function describeBadNumber(fullText) {
|
|
|
1107
1437
|
return `${RADIX_ARTICLE[radix]} ${name} integer cannot have a sign, because it states a bit pattern rather than a quantity`;
|
|
1108
1438
|
}
|
|
1109
1439
|
if (digits === '') {
|
|
1110
|
-
return `Expected ${name} digits after
|
|
1440
|
+
return `Expected ${name} digits after “0${radix}”`;
|
|
1111
1441
|
}
|
|
1112
1442
|
if (radix === 'x' && /[a-f]/v.test(digits) && !/[^\da-f_]/iv.test(digits)) {
|
|
1113
|
-
|
|
1443
|
+
// The uppercase spelling is only suggested when it is valid, so a misplaced underscore is reported first, and a value outside the 64-bit range gets no example.
|
|
1444
|
+
if (!/^[\da-f]+(?:_[\da-f]+)*$/iv.test(digits)) {
|
|
1445
|
+
return 'An underscore in a number must be between two digits';
|
|
1446
|
+
}
|
|
1447
|
+
const uppercase = `0x${digits.toUpperCase()}`;
|
|
1448
|
+
return isValidValue(uppercase) ? `Hexadecimal digits are uppercase: ${abbreviate(uppercase)}` : 'Hexadecimal digits are uppercase';
|
|
1114
1449
|
}
|
|
1115
|
-
|
|
1450
|
+
// A lowercase hexadecimal digit is a digit in the wrong case, so the one named is a character that is no digit in either case.
|
|
1451
|
+
const validCharacter = { x: /[\da-f_]/iv, o: /[0-7_]/v, b: /[01_]/v }[radix];
|
|
1116
1452
|
const character = [...digits].find(character => !validCharacter.test(character));
|
|
1117
|
-
return character === undefined ? 'An underscore in a number must be between two digits' : `Invalid ${name} digit
|
|
1453
|
+
return character === undefined ? 'An underscore in a number must be between two digits' : `Invalid ${name} digit “${character}”`;
|
|
1118
1454
|
}
|
|
1119
1455
|
if (/^-?\d[\d_]*E/v.test(text) || /^-?\d[\d_]*\.[\d_]+E/v.test(text)) {
|
|
1120
|
-
return 'An exponent marker is a lowercase
|
|
1456
|
+
return 'An exponent marker is a lowercase “e”';
|
|
1121
1457
|
}
|
|
1122
1458
|
if (/^-?[\d_]+\.(?:$|[^\d_])/v.test(text)) {
|
|
1123
1459
|
return 'A decimal point must be followed by a digit';
|
|
1124
1460
|
}
|
|
1125
1461
|
if (text.startsWith('-.')) {
|
|
1126
|
-
return 'A number cannot begin with
|
|
1462
|
+
return 'A number cannot begin with “.”; write a digit before it, as in -0.5';
|
|
1463
|
+
}
|
|
1464
|
+
if (/^\d+(?::\d+)+$/v.test(text)) {
|
|
1465
|
+
return /^\d{1,2}:\d{2}(?::\d{2}(?:\.\d+)?)?$/v.test(text) ? 'A time of day is a string, so it must be quoted' : `Invalid number “${abbreviate(text)}”. A value that contains “:” must be quoted, as in '${abbreviate(text)}'`;
|
|
1127
1466
|
}
|
|
1128
1467
|
if (/^-?0[\d_]/v.test(text)) {
|
|
1129
|
-
|
|
1468
|
+
// Removing the zero gives a valid number with a different meaning, so the message says what the zero usually meant.
|
|
1469
|
+
// A `'...'` string cannot hold a `'`, so then there is no example.
|
|
1470
|
+
const identifier = `an identifier, such as a ZIP code, as a string${unquotedText.includes('\'') ? '' : `: '${abbreviate(unquotedText)}'`}`;
|
|
1471
|
+
// The octal suggestion is only for the number on its own, not for one that more text follows, as in `0412 345 678`, and only when it is in range. It is decided on the whole number, because a cut can end before a digit that is not octal or that changes the value.
|
|
1472
|
+
const digits = fullText.replace(/^0+/v, '');
|
|
1473
|
+
const octal = `0o${digits === '' ? '0' : digits}`;
|
|
1474
|
+
if (unquotedText === fullText && /^0[0-7]+$/v.test(fullText) && isValidValue(octal)) {
|
|
1475
|
+
return `Leading zeros are not allowed in a decimal number. Write an octal number, such as a file mode, as ${abbreviate(octal)}, and ${identifier}`;
|
|
1476
|
+
}
|
|
1477
|
+
return /^0\d+$/v.test(text) ? `Leading zeros are not allowed in a decimal number. Write ${identifier}` : 'Leading zeros are not allowed in a decimal number';
|
|
1130
1478
|
}
|
|
1131
1479
|
if (/_(?:$|\D)|(?:^|\D)_/v.test(text)) {
|
|
1132
1480
|
return 'An underscore in a number must be between two digits';
|
|
@@ -1135,37 +1483,65 @@ function describeBadNumber(fullText) {
|
|
|
1135
1483
|
return 'Leading zeros are not allowed in an exponent';
|
|
1136
1484
|
}
|
|
1137
1485
|
if (/^-?\d[\d_]*(?:\.[\d_]+)?e-0$/v.test(text)) {
|
|
1138
|
-
return '
|
|
1486
|
+
return '“e-0” is not allowed, because an exponent of zero has one spelling: e0';
|
|
1139
1487
|
}
|
|
1140
1488
|
if (/e-?$/v.test(text)) {
|
|
1141
|
-
return 'Expected digits after the exponent marker
|
|
1489
|
+
return 'Expected digits after the exponent marker “e”';
|
|
1142
1490
|
}
|
|
1143
1491
|
if (text.split('.').length > 2) {
|
|
1144
|
-
return `Invalid number
|
|
1492
|
+
return `Invalid number “${abbreviate(text)}”. A value with several dots, such as a version number, must be quoted`;
|
|
1145
1493
|
}
|
|
1146
1494
|
if (/^-(?:\D|$)/v.test(text)) {
|
|
1147
1495
|
if (/^-nan/iv.test(text)) {
|
|
1148
1496
|
return 'NaN is not representable. Use null for a missing value';
|
|
1149
1497
|
}
|
|
1150
|
-
return /^-inf/iv.test(text) ?
|
|
1498
|
+
return /^-inf/iv.test(text) ? `“${abbreviate(text)}” is not a value. Negative infinity is written -infinity` : 'Expected a digit or “infinity” after “-”';
|
|
1151
1499
|
}
|
|
1152
|
-
return /[A-Za-z]/v.test(text) ? `Invalid number
|
|
1500
|
+
return /[A-Za-z]/v.test(text) ? `Invalid number “${abbreviate(text)}”. A string value must be quoted${quotingExample(unquotedText)}` : `Invalid number “${abbreviate(text)}”`;
|
|
1153
1501
|
}
|
|
1154
|
-
|
|
1502
|
+
/*
|
|
1503
|
+
A date alone, which is not an instant. The instant it could be is only shown when the date exists, so that the example is valid.
|
|
1504
|
+
*/
|
|
1505
|
+
function describeDate(date) {
|
|
1506
|
+
const [year, month, day] = date.split('-').map(Number);
|
|
1507
|
+
const isExisting = year >= 1 && month >= 1 && month <= 12 && day >= 1 && day <= daysInMonth(year, month);
|
|
1508
|
+
return `${date} is a date, not an instant. Write a date as a string, as in '${date}'${isExisting ? `. An instant needs a time and an offset, as in ${date}T00:00:00Z` : ''}`;
|
|
1509
|
+
}
|
|
1510
|
+
/*
|
|
1511
|
+
`time` is the token after a space that follows `text`, or an empty string. `isWholeValue` is whether nothing that may be part of the instant follows `text`.
|
|
1512
|
+
*/
|
|
1513
|
+
function describeBadInstant(text, time, isWholeValue) {
|
|
1155
1514
|
if (text.length > MAX_DIAGNOSED_LENGTH) {
|
|
1156
|
-
return `Invalid instant
|
|
1515
|
+
return `Invalid instant “${abbreviate(text)}”. ${INSTANT_FORMAT}`;
|
|
1157
1516
|
}
|
|
1158
1517
|
if (/^\d{4}-\d{2}-\d{2}$/v.test(text)) {
|
|
1159
|
-
return
|
|
1518
|
+
return /^\d{2}:\d{2}/v.test(time) ? describeSpaceSeparatedInstant(text, time, isWholeValue) : describeDate(text);
|
|
1160
1519
|
}
|
|
1161
1520
|
if (/^\d{4}-\d{2}-\d{2}t/v.test(text)) {
|
|
1162
|
-
return 'The date and time separator in an instant is an uppercase
|
|
1521
|
+
return 'The date and time separator in an instant is an uppercase “T”';
|
|
1163
1522
|
}
|
|
1164
1523
|
if (text.endsWith('z')) {
|
|
1165
|
-
return 'The UTC offset in an instant is an uppercase
|
|
1524
|
+
return 'The UTC offset in an instant is an uppercase “Z”';
|
|
1166
1525
|
}
|
|
1167
1526
|
if (/[+\-]\d{4}$/v.test(text)) {
|
|
1168
1527
|
return 'An instant\'s offset is written with a colon, as in +07:00';
|
|
1169
1528
|
}
|
|
1170
|
-
|
|
1529
|
+
if (!LOCAL_DATE_TIME.test(text)) {
|
|
1530
|
+
return `Invalid instant “${abbreviate(text)}”. ${INSTANT_FORMAT}`;
|
|
1531
|
+
}
|
|
1532
|
+
return isWholeValue ? `An instant needs an offset: Z or ±HH:MM. A date and time without an offset is a local time, which is not an instant, so write it as a string, as in '${abbreviate(text)}', or add the offset it was meant in` : 'An instant needs an offset: Z or ±HH:MM';
|
|
1533
|
+
}
|
|
1534
|
+
/*
|
|
1535
|
+
A date and a time with a space between them, where an instant has a "T". Following the example must not turn a local time into UTC silently, so a time without an offset is described as what it is. An instant is only shown when it is valid, so a date that does not exist, a time out of range, or an instant outside the years 0001 to 9999 in UTC gets the general format instead.
|
|
1536
|
+
*/
|
|
1537
|
+
function describeSpaceSeparatedInstant(date, time, isWholeValue) {
|
|
1538
|
+
const separator = 'The date and time separator in an instant is an uppercase “T”, not a space';
|
|
1539
|
+
const instant = `${date}T${time}`;
|
|
1540
|
+
if (instant.length > MAX_DIAGNOSED_LENGTH) {
|
|
1541
|
+
return `${separator}. ${INSTANT_FORMAT}`;
|
|
1542
|
+
}
|
|
1543
|
+
if (isValidValue(instant)) {
|
|
1544
|
+
return `${separator}, as in ${abbreviate(instant)}`;
|
|
1545
|
+
}
|
|
1546
|
+
return isWholeValue && LOCAL_DATE_TIME.test(instant) && isValidValue(`${instant}Z`) ? `${separator}, and an instant needs the offset it was meant in, as in ${abbreviate(instant)}Z for UTC. A date and time without an offset is a local time, which is not an instant, so write it as a string, as in '${abbreviate(`${date} ${time}`)}'` : `${separator}. ${INSTANT_FORMAT}`;
|
|
1171
1547
|
}
|