soml-lang 0.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/index.d.ts +151 -0
- package/index.js +3 -0
- package/license +9 -0
- package/package.json +91 -0
- package/readme.md +224 -0
- package/source/error.js +130 -0
- package/source/parse.js +1574 -0
- package/source/shared.js +166 -0
- package/source/stringify.js +439 -0
package/source/parse.js
ADDED
|
@@ -0,0 +1,1574 @@
|
|
|
1
|
+
import {ParseError} from './error.js';
|
|
2
|
+
import {
|
|
3
|
+
MAX_DEPTH,
|
|
4
|
+
INT64_MIN,
|
|
5
|
+
INT64_MAX,
|
|
6
|
+
MIN_INSTANT,
|
|
7
|
+
MAX_INSTANT,
|
|
8
|
+
DURATION_UNITS,
|
|
9
|
+
createDuration,
|
|
10
|
+
trimTrailingZeros,
|
|
11
|
+
validateIntegersOption,
|
|
12
|
+
describeCharacter,
|
|
13
|
+
formatCodePoint,
|
|
14
|
+
isBareKey,
|
|
15
|
+
bareKeyCharacters,
|
|
16
|
+
abbreviate,
|
|
17
|
+
} from './shared.js';
|
|
18
|
+
|
|
19
|
+
const TAB = 0x09;
|
|
20
|
+
const LF = 0x0A;
|
|
21
|
+
const SPACE = 0x20;
|
|
22
|
+
const DOUBLE_QUOTE = 0x22;
|
|
23
|
+
const HASH = 0x23;
|
|
24
|
+
const SINGLE_QUOTE = 0x27;
|
|
25
|
+
const ASTERISK = 0x2A;
|
|
26
|
+
const COMMA = 0x2C;
|
|
27
|
+
const DASH = 0x2D;
|
|
28
|
+
const DOT = 0x2E;
|
|
29
|
+
const SLASH = 0x2F;
|
|
30
|
+
const COLON = 0x3A;
|
|
31
|
+
const OPEN_BRACKET = 0x5B;
|
|
32
|
+
const CLOSE_BRACKET = 0x5D;
|
|
33
|
+
const OPEN_BRACE = 0x7B;
|
|
34
|
+
const CLOSE_BRACE = 0x7D;
|
|
35
|
+
|
|
36
|
+
/*
|
|
37
|
+
What may directly follow a scalar value: whitespace, a separator, a closing bracket, a comment, or the end.
|
|
38
|
+
*/
|
|
39
|
+
const valueEndCharacters = new Uint8Array(128);
|
|
40
|
+
|
|
41
|
+
for (const character of ' \t\n,]}#/') {
|
|
42
|
+
valueEndCharacters[character.codePointAt(0)] = 1;
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
/*
|
|
46
|
+
A number, an instant, or a duration is lexed as one maximal run of these characters, and then validated as a whole. The validation is a hand-written scan rather than a regular expression, because a regular expression with a repeated group runs out of stack on a token of a few million characters.
|
|
47
|
+
*/
|
|
48
|
+
const numberCharacters = new Uint8Array(128);
|
|
49
|
+
|
|
50
|
+
for (const character of 'ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789_.:+-') {
|
|
51
|
+
numberCharacters[character.codePointAt(0)] = 1;
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
const RADIX_DIGIT = {
|
|
55
|
+
x: code => (code >= 0x30 && code <= 0x39) || (code >= 0x41 && code <= 0x46),
|
|
56
|
+
o: code => code >= 0x30 && code <= 0x37,
|
|
57
|
+
b: code => code === 0x30 || code === 0x31,
|
|
58
|
+
};
|
|
59
|
+
|
|
60
|
+
/*
|
|
61
|
+
The longest token the diagnostic regular expressions look at. Their repeated groups need stack in proportion to the input, so a longer invalid token is described from its first part.
|
|
62
|
+
*/
|
|
63
|
+
const MAX_DIAGNOSED_LENGTH = 1000;
|
|
64
|
+
|
|
65
|
+
const RADIX_NAME = {x: 'hexadecimal', o: 'octal', b: 'binary'};
|
|
66
|
+
const RADIX_ARTICLE = {x: 'A', o: 'An', b: 'A'};
|
|
67
|
+
const INSTANT_PREFIX = /^\d{4}-\d{2}-\d{2}/v;
|
|
68
|
+
const DURATION_UNIT_NAMES = DURATION_UNITS.keys().toArray();
|
|
69
|
+
const INSTANT = /^(?<year>\d{4})-(?<month>\d{2})-(?<day>\d{2})T(?<hour>\d{2}):(?<minute>\d{2}):(?<second>\d{2})(?:\.(?<fraction>\d+))?(?<offset>Z|[+\-](?<offsetHour>\d{2}):(?<offsetMinute>\d{2}))$/v;
|
|
70
|
+
const UNICODE_ESCAPE = /u\{(?<hex>[\da-f]{1,6})\}/vy;
|
|
71
|
+
/*
|
|
72
|
+
Every C0 control character except tab and line feed, and DEL. They are errors anywhere in a document.
|
|
73
|
+
*/
|
|
74
|
+
// eslint-disable-next-line no-control-regex, regexp/no-control-character -- Control characters are exactly what must be found.
|
|
75
|
+
const CONTROL_CHARACTER = /[\u{0}-\u{8}\u{B}-\u{1F}\u{7F}]/v;
|
|
76
|
+
const LITERAL_STRING_END = /[\n']/gv;
|
|
77
|
+
const BLANK_LINE = /^[\t ]*$/v;
|
|
78
|
+
const ESCAPED_STRING_SPECIAL = /[\n"\\]/gv;
|
|
79
|
+
|
|
80
|
+
const SIMPLE_ESCAPES = {
|
|
81
|
+
'\\': '\\',
|
|
82
|
+
'"': '"',
|
|
83
|
+
n: '\n',
|
|
84
|
+
t: '\t',
|
|
85
|
+
};
|
|
86
|
+
|
|
87
|
+
/**
|
|
88
|
+
Parse a document.
|
|
89
|
+
|
|
90
|
+
@param {string | Uint8Array} text
|
|
91
|
+
@param {{integers?: 'bigint' | 'number'}} [options]
|
|
92
|
+
*/
|
|
93
|
+
export function parse(text, options = {}) {
|
|
94
|
+
const {integers = 'bigint'} = options;
|
|
95
|
+
validateIntegersOption(integers);
|
|
96
|
+
const source = decode(text);
|
|
97
|
+
checkCharacters(source);
|
|
98
|
+
return new Parser(source, integers).parseDocument();
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
const typedArrayTag = Object.getOwnPropertyDescriptor(Object.getPrototypeOf(Uint8Array.prototype), Symbol.toStringTag).get;
|
|
102
|
+
|
|
103
|
+
function decode(text) {
|
|
104
|
+
if (typeof text === 'string') {
|
|
105
|
+
return text;
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
// A brand check rather than `instanceof`, so that bytes from another realm, such as a `vm` context, are accepted too. The typed array tag getter reads an internal slot, so it cannot be faked with `Symbol.toStringTag`.
|
|
109
|
+
if (typedArrayTag.call(text) === 'Uint8Array') {
|
|
110
|
+
try {
|
|
111
|
+
// `ignoreBOM` keeps a BOM in the output, so that it can be rejected rather than silently dropped.
|
|
112
|
+
return new TextDecoder('utf-8', {fatal: true, ignoreBOM: true}).decode(text);
|
|
113
|
+
} catch (error) {
|
|
114
|
+
const offset = findInvalidUtf8(text);
|
|
115
|
+
|
|
116
|
+
// Every byte is valid, so the decoder failed for another reason, such as input longer than the longest possible string.
|
|
117
|
+
if (offset === text.length) {
|
|
118
|
+
throw error;
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
const valid = new TextDecoder('utf-8', {ignoreBOM: true}).decode(text.subarray(0, offset));
|
|
122
|
+
|
|
123
|
+
// A sequence that the end of the input cuts short decodes without error when more bytes may follow.
|
|
124
|
+
if (isTruncatedUtf8(text.subarray(offset))) {
|
|
125
|
+
throw new ParseError('Incomplete UTF-8 sequence at the end of the input', valid, valid.length);
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
throw new ParseError(`Invalid UTF-8 byte 0x${text[offset].toString(16).toUpperCase().padStart(2, '0')}`, valid, valid.length);
|
|
129
|
+
}
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
throw new TypeError(`Expected a string or a Uint8Array, got ${typeof text === 'object' ? (text === null ? 'null' : 'an object') : typeof text}`);
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
/*
|
|
136
|
+
Finds the first byte of the first invalid sequence. Only called on the error path, so each multi-byte sequence is checked with a fatal decoder rather than by reimplementing the UTF-8 rules for overlong forms, surrogates, and the U+10FFFF limit.
|
|
137
|
+
*/
|
|
138
|
+
function findInvalidUtf8(bytes) {
|
|
139
|
+
const decoder = new TextDecoder('utf-8', {fatal: true, ignoreBOM: true});
|
|
140
|
+
let index = 0;
|
|
141
|
+
|
|
142
|
+
while (index < bytes.length) {
|
|
143
|
+
const byte = bytes[index];
|
|
144
|
+
|
|
145
|
+
if (byte < 0x80) {
|
|
146
|
+
index++;
|
|
147
|
+
continue;
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
const length = byte >= 0xF0 ? 4 : (byte >= 0xE0 ? 3 : 2);
|
|
151
|
+
const sequence = bytes.subarray(index, index + length);
|
|
152
|
+
|
|
153
|
+
// A fatal decode also throws for a sequence that the end of the input cuts short.
|
|
154
|
+
try {
|
|
155
|
+
decoder.decode(sequence);
|
|
156
|
+
} catch {
|
|
157
|
+
return index;
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
index += length;
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
return index;
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
function isTruncatedUtf8(bytes) {
|
|
167
|
+
try {
|
|
168
|
+
new TextDecoder('utf-8', {fatal: true}).decode(bytes, {stream: true});
|
|
169
|
+
return true;
|
|
170
|
+
} catch {
|
|
171
|
+
return false;
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
/*
|
|
176
|
+
Characters that are errors wherever they appear, so they are checked once, up front.
|
|
177
|
+
*/
|
|
178
|
+
function checkCharacters(source) {
|
|
179
|
+
if (source.charCodeAt(0) === 0xFE_FF) {
|
|
180
|
+
throw new ParseError('A byte order mark (BOM) is not allowed', source, 0);
|
|
181
|
+
}
|
|
182
|
+
|
|
183
|
+
const control = CONTROL_CHARACTER.exec(source);
|
|
184
|
+
|
|
185
|
+
if (control !== null) {
|
|
186
|
+
const {index} = control;
|
|
187
|
+
const code = source.charCodeAt(index);
|
|
188
|
+
|
|
189
|
+
if (code === 0x0D) {
|
|
190
|
+
throw new ParseError('A carriage return (U+000D) is not allowed anywhere. Use LF line endings', source, index);
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
const escape = String.raw`\u{${code.toString(16)}}`;
|
|
194
|
+
throw new ParseError(`A raw control character (${formatCodePoint(code)}) is not allowed anywhere, including in strings and comments. In a string, write it as the escape ${escape} inside "..."`, source, index);
|
|
195
|
+
}
|
|
196
|
+
|
|
197
|
+
if (!source.isWellFormed()) {
|
|
198
|
+
// With the `v` flag, a surrogate in the class matches only when it is unpaired.
|
|
199
|
+
const offset = /[\u{D800}-\u{DFFF}]/v.exec(source).index;
|
|
200
|
+
throw new ParseError(`A lone surrogate (${describeCharacter(source.charCodeAt(offset))}) is not a Unicode scalar value`, source, offset);
|
|
201
|
+
}
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
/*
|
|
205
|
+
Creates an own property without ever invoking a setter, like `JSON.parse`. Plain assignment is used when `Object.prototype` has no property of that name, which is the fast common case. Otherwise, as for `__proto__`, `toString`, or anything a library added to the prototype, assignment would call a setter or fail on a frozen prototype.
|
|
206
|
+
*/
|
|
207
|
+
function defineMember(object, key, value) {
|
|
208
|
+
if (Reflect.has(Object.prototype, key)) {
|
|
209
|
+
Object.defineProperty(object, key, {
|
|
210
|
+
value,
|
|
211
|
+
writable: true,
|
|
212
|
+
enumerable: true,
|
|
213
|
+
configurable: true,
|
|
214
|
+
});
|
|
215
|
+
} else {
|
|
216
|
+
object[key] = value;
|
|
217
|
+
}
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
function describeValue(value) {
|
|
221
|
+
if (Array.isArray(value)) {
|
|
222
|
+
return 'an array';
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
return value !== null && typeof value === 'object' && !(value instanceof Temporal.Instant) && !(value instanceof Temporal.Duration) ? 'an object' : 'a value';
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
function formatPath(path) {
|
|
229
|
+
return path.map(segment => isBareKey(segment) ? abbreviate(segment) : JSON.stringify(abbreviate(segment))).join('.');
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
class Parser {
|
|
233
|
+
#source;
|
|
234
|
+
#index = 0;
|
|
235
|
+
#integers;
|
|
236
|
+
/*
|
|
237
|
+
Objects created by dotted keys. They may be extended by further dotted keys, while an object written with braces is closed.
|
|
238
|
+
*/
|
|
239
|
+
#dottedObjects = new WeakSet();
|
|
240
|
+
|
|
241
|
+
constructor(source, integers) {
|
|
242
|
+
this.#source = source;
|
|
243
|
+
this.#integers = integers;
|
|
244
|
+
}
|
|
245
|
+
|
|
246
|
+
#fail(reason, offset = this.#index) {
|
|
247
|
+
throw new ParseError(reason, this.#source, offset);
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
#code(offset = this.#index) {
|
|
251
|
+
return this.#source.charCodeAt(offset);
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
#isAtEnd() {
|
|
255
|
+
return this.#index >= this.#source.length;
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
parseDocument() {
|
|
259
|
+
this.#skipTrivia();
|
|
260
|
+
|
|
261
|
+
if (this.#isAtEnd()) {
|
|
262
|
+
this.#fail('A document must contain an object or an array, but this one is empty');
|
|
263
|
+
}
|
|
264
|
+
|
|
265
|
+
const code = this.#code();
|
|
266
|
+
let value;
|
|
267
|
+
|
|
268
|
+
if (code === OPEN_BRACE) {
|
|
269
|
+
value = this.#parseObject(1);
|
|
270
|
+
} else if (code === OPEN_BRACKET) {
|
|
271
|
+
value = this.#parseArray(1);
|
|
272
|
+
} else {
|
|
273
|
+
value = this.#parseBareObject();
|
|
274
|
+
}
|
|
275
|
+
|
|
276
|
+
this.#skipTrivia();
|
|
277
|
+
|
|
278
|
+
if (!this.#isAtEnd()) {
|
|
279
|
+
this.#fail(`Unexpected ${this.#describeHere()} after the end of the document`);
|
|
280
|
+
}
|
|
281
|
+
|
|
282
|
+
return value;
|
|
283
|
+
}
|
|
284
|
+
|
|
285
|
+
/*
|
|
286
|
+
`bare-object = removable entry ( entry-sep removable entry )*`, where a separator is one or more line breaks.
|
|
287
|
+
*/
|
|
288
|
+
#parseBareObject() {
|
|
289
|
+
const object = {};
|
|
290
|
+
const start = this.#index;
|
|
291
|
+
|
|
292
|
+
try {
|
|
293
|
+
this.#parseEntry(object, 1);
|
|
294
|
+
} catch (error) {
|
|
295
|
+
if (error instanceof ParseError) {
|
|
296
|
+
this.#diagnoseBareValue(start);
|
|
297
|
+
}
|
|
298
|
+
|
|
299
|
+
throw error;
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
for (;;) {
|
|
303
|
+
const sawLineBreak = this.#skipTrivia();
|
|
304
|
+
|
|
305
|
+
if (this.#isAtEnd()) {
|
|
306
|
+
return object;
|
|
307
|
+
}
|
|
308
|
+
|
|
309
|
+
if (!sawLineBreak) {
|
|
310
|
+
if (this.#code() === COMMA) {
|
|
311
|
+
this.#fail('Top-level entries are separated by line breaks, not commas. Use braces for a one-line object');
|
|
312
|
+
}
|
|
313
|
+
|
|
314
|
+
this.#fail(`Expected a line break before the next entry, but found ${this.#describeHere()}`);
|
|
315
|
+
}
|
|
316
|
+
|
|
317
|
+
this.#parseEntry(object, 1);
|
|
318
|
+
}
|
|
319
|
+
}
|
|
320
|
+
|
|
321
|
+
/*
|
|
322
|
+
A whole document that is one scalar, such as `5` or an instant, fails as an entry. That failure is reported as what it is.
|
|
323
|
+
*/
|
|
324
|
+
#diagnoseBareValue(start) {
|
|
325
|
+
this.#index = start;
|
|
326
|
+
let isBareValue = false;
|
|
327
|
+
|
|
328
|
+
try {
|
|
329
|
+
this.#parseValue(1);
|
|
330
|
+
this.#skipTrivia();
|
|
331
|
+
isBareValue = this.#isAtEnd();
|
|
332
|
+
} catch {
|
|
333
|
+
// Not a bare value either, so the original error stands.
|
|
334
|
+
}
|
|
335
|
+
|
|
336
|
+
if (isBareValue) {
|
|
337
|
+
this.#fail('A bare value is not a document. A document is an object or an array, so write it as `key: value` or `[value]`', start);
|
|
338
|
+
}
|
|
339
|
+
}
|
|
340
|
+
|
|
341
|
+
/*
|
|
342
|
+
Skips spaces, tabs, line breaks, and comments, and reports whether a line break was crossed. A line break inside a block comment does not count.
|
|
343
|
+
*/
|
|
344
|
+
#skipTrivia() {
|
|
345
|
+
const source = this.#source;
|
|
346
|
+
let hasCrossedLineBreak = false;
|
|
347
|
+
|
|
348
|
+
for (;;) {
|
|
349
|
+
switch (source.charCodeAt(this.#index)) {
|
|
350
|
+
case SPACE:
|
|
351
|
+
case TAB: {
|
|
352
|
+
this.#index++;
|
|
353
|
+
break;
|
|
354
|
+
}
|
|
355
|
+
|
|
356
|
+
case LF: {
|
|
357
|
+
hasCrossedLineBreak = true;
|
|
358
|
+
this.#index++;
|
|
359
|
+
break;
|
|
360
|
+
}
|
|
361
|
+
|
|
362
|
+
case HASH: {
|
|
363
|
+
const end = source.indexOf('\n', this.#index);
|
|
364
|
+
this.#index = end === -1 ? source.length : end;
|
|
365
|
+
break;
|
|
366
|
+
}
|
|
367
|
+
|
|
368
|
+
case SLASH: {
|
|
369
|
+
if (source.charCodeAt(this.#index + 1) !== ASTERISK) {
|
|
370
|
+
return hasCrossedLineBreak;
|
|
371
|
+
}
|
|
372
|
+
|
|
373
|
+
this.#skipBlockComment();
|
|
374
|
+
break;
|
|
375
|
+
}
|
|
376
|
+
|
|
377
|
+
default: {
|
|
378
|
+
return hasCrossedLineBreak;
|
|
379
|
+
}
|
|
380
|
+
}
|
|
381
|
+
}
|
|
382
|
+
}
|
|
383
|
+
|
|
384
|
+
#skipBlockComment() {
|
|
385
|
+
const source = this.#source;
|
|
386
|
+
const start = this.#index;
|
|
387
|
+
const end = source.indexOf('*/', start + 2);
|
|
388
|
+
|
|
389
|
+
if (end === -1) {
|
|
390
|
+
this.#fail('Unterminated block comment', start);
|
|
391
|
+
}
|
|
392
|
+
|
|
393
|
+
const nested = source.indexOf('/*', start + 2);
|
|
394
|
+
|
|
395
|
+
// The body may not contain `/*`. An opening that overlaps the closing `*/`, as in `/*/`, is not inside the body.
|
|
396
|
+
if (nested !== -1 && nested + 2 <= end) {
|
|
397
|
+
this.#fail('Block comments cannot be nested, and their body may not contain "/*"', nested);
|
|
398
|
+
}
|
|
399
|
+
|
|
400
|
+
this.#index = end + 2;
|
|
401
|
+
}
|
|
402
|
+
|
|
403
|
+
/*
|
|
404
|
+
Consumes `/-` and the spaces and tabs after it, and reports whether it was there.
|
|
405
|
+
*/
|
|
406
|
+
#parseRemoval() {
|
|
407
|
+
if (this.#code() !== SLASH || this.#code(this.#index + 1) !== DASH) {
|
|
408
|
+
return false;
|
|
409
|
+
}
|
|
410
|
+
|
|
411
|
+
const start = this.#index;
|
|
412
|
+
this.#index += 2;
|
|
413
|
+
|
|
414
|
+
while (this.#code() === SPACE || this.#code() === TAB) {
|
|
415
|
+
this.#index++;
|
|
416
|
+
}
|
|
417
|
+
|
|
418
|
+
const code = this.#code();
|
|
419
|
+
|
|
420
|
+
if (this.#isAtEnd() || code === LF || code === HASH || (code === SLASH && this.#code(this.#index + 1) === ASTERISK) || code === COMMA || code === CLOSE_BRACE || code === CLOSE_BRACKET) {
|
|
421
|
+
this.#fail('"/-" must be followed by the element it removes, on the same line', start);
|
|
422
|
+
}
|
|
423
|
+
|
|
424
|
+
return true;
|
|
425
|
+
}
|
|
426
|
+
|
|
427
|
+
/*
|
|
428
|
+
`entry = key ":" ws value`, with an optional `/-` before it.
|
|
429
|
+
*/
|
|
430
|
+
#parseEntry(object, depth) {
|
|
431
|
+
const isRemoved = this.#parseRemoval();
|
|
432
|
+
const keyStart = this.#index;
|
|
433
|
+
const path = this.#parseKey();
|
|
434
|
+
|
|
435
|
+
if (depth + path.length - 1 > MAX_DEPTH) {
|
|
436
|
+
this.#fail(`The document is nested more than ${MAX_DEPTH} levels deep`, keyStart);
|
|
437
|
+
}
|
|
438
|
+
|
|
439
|
+
const code = this.#code();
|
|
440
|
+
|
|
441
|
+
if (code !== COLON) {
|
|
442
|
+
this.#failMissingColon(keyStart);
|
|
443
|
+
}
|
|
444
|
+
|
|
445
|
+
this.#index++;
|
|
446
|
+
this.#skipTrivia();
|
|
447
|
+
const value = this.#parseValue(depth + path.length);
|
|
448
|
+
|
|
449
|
+
if (!isRemoved) {
|
|
450
|
+
this.#assign(object, path, value, keyStart);
|
|
451
|
+
}
|
|
452
|
+
}
|
|
453
|
+
|
|
454
|
+
#failMissingColon(keyStart) {
|
|
455
|
+
const source = this.#source;
|
|
456
|
+
let next = this.#index;
|
|
457
|
+
|
|
458
|
+
while (source.charCodeAt(next) === SPACE || source.charCodeAt(next) === TAB) {
|
|
459
|
+
next++;
|
|
460
|
+
}
|
|
461
|
+
|
|
462
|
+
const nextCode = source.charCodeAt(next);
|
|
463
|
+
|
|
464
|
+
if (next > this.#index && nextCode === COLON) {
|
|
465
|
+
this.#fail('Whitespace is not allowed between a key and its ":"');
|
|
466
|
+
}
|
|
467
|
+
|
|
468
|
+
const colon = source.indexOf(':', next);
|
|
469
|
+
|
|
470
|
+
// A key with a space in it, such as `the name: 1`. The hint is only given for plain words, because quoting a dotted or quoted key would change what it means.
|
|
471
|
+
if (next > this.#index && colon !== -1 && colon < findLineEnd(source, next) && nextCode < 128 && bareKeyCharacters[nextCode] === 1) {
|
|
472
|
+
const key = source.slice(keyStart, colon).trimEnd();
|
|
473
|
+
|
|
474
|
+
if (isWordsWithSpaces(key)) {
|
|
475
|
+
this.#fail(`A bare key cannot contain spaces. Quote it, as in '${abbreviate(key)}'`, keyStart);
|
|
476
|
+
}
|
|
477
|
+
}
|
|
478
|
+
|
|
479
|
+
if (this.#isAtEnd() || this.#code() === LF) {
|
|
480
|
+
this.#fail('Expected ":" after the key');
|
|
481
|
+
}
|
|
482
|
+
|
|
483
|
+
// A character directly after the key is most likely meant to be part of it.
|
|
484
|
+
const hint = next === this.#index ? '. A key that contains characters other than letters, digits, "_", and "-" must be quoted' : '';
|
|
485
|
+
this.#index = next;
|
|
486
|
+
this.#fail(`Expected ":" after the key, but found ${this.#describeHere()}${hint}`);
|
|
487
|
+
}
|
|
488
|
+
|
|
489
|
+
#parseKey() {
|
|
490
|
+
const path = [this.#parseKeySegment()];
|
|
491
|
+
|
|
492
|
+
while (this.#code() === DOT) {
|
|
493
|
+
this.#index++;
|
|
494
|
+
path.push(this.#parseKeySegment(true));
|
|
495
|
+
}
|
|
496
|
+
|
|
497
|
+
return path;
|
|
498
|
+
}
|
|
499
|
+
|
|
500
|
+
#parseKeySegment(isAfterDot = false) {
|
|
501
|
+
const source = this.#source;
|
|
502
|
+
const start = this.#index;
|
|
503
|
+
const code = source.charCodeAt(start);
|
|
504
|
+
|
|
505
|
+
if (code === SINGLE_QUOTE || code === DOUBLE_QUOTE) {
|
|
506
|
+
if (source.charCodeAt(start + 1) === code && source.charCodeAt(start + 2) === code) {
|
|
507
|
+
this.#fail('A block string cannot be a key');
|
|
508
|
+
}
|
|
509
|
+
|
|
510
|
+
// A key is a string, so it may hold anything a string can, including a line feed written as an escape.
|
|
511
|
+
return code === SINGLE_QUOTE ? this.#parseLiteralString() : this.#parseEscapedString();
|
|
512
|
+
}
|
|
513
|
+
|
|
514
|
+
let end = start;
|
|
515
|
+
|
|
516
|
+
while (end < source.length && source.charCodeAt(end) < 128 && bareKeyCharacters[source.charCodeAt(end)] === 1) {
|
|
517
|
+
end++;
|
|
518
|
+
}
|
|
519
|
+
|
|
520
|
+
if (end === start) {
|
|
521
|
+
if (isAfterDot) {
|
|
522
|
+
this.#fail(this.#isAtEnd() ? 'Expected a key segment after "."' : `Expected a key segment after ".", but found ${this.#describeHere()}`);
|
|
523
|
+
}
|
|
524
|
+
|
|
525
|
+
this.#fail(this.#isAtEnd() ? 'Expected a key' : `Expected a key, but found ${this.#describeHere()}`);
|
|
526
|
+
}
|
|
527
|
+
|
|
528
|
+
this.#index = end;
|
|
529
|
+
return source.slice(start, end);
|
|
530
|
+
}
|
|
531
|
+
|
|
532
|
+
#assign(object, path, value, keyStart) {
|
|
533
|
+
let target = object;
|
|
534
|
+
const lastIndex = path.length - 1;
|
|
535
|
+
|
|
536
|
+
for (let index = 0; index < lastIndex; index++) {
|
|
537
|
+
const key = path[index];
|
|
538
|
+
|
|
539
|
+
if (!Object.hasOwn(target, key)) {
|
|
540
|
+
const child = {};
|
|
541
|
+
this.#dottedObjects.add(child);
|
|
542
|
+
defineMember(target, key, child);
|
|
543
|
+
target = child;
|
|
544
|
+
continue;
|
|
545
|
+
}
|
|
546
|
+
|
|
547
|
+
const existing = target[key];
|
|
548
|
+
|
|
549
|
+
if (!this.#dottedObjects.has(existing)) {
|
|
550
|
+
const prefix = formatPath(path.slice(0, index + 1));
|
|
551
|
+
const what = describeValue(existing);
|
|
552
|
+
const reason = what === 'an object' ? 'an object written with braces, which is closed' : what;
|
|
553
|
+
this.#fail(`Cannot set ${formatPath(path)}, because ${prefix} is already ${reason} and a dotted key cannot extend it`, keyStart);
|
|
554
|
+
}
|
|
555
|
+
|
|
556
|
+
target = existing;
|
|
557
|
+
}
|
|
558
|
+
|
|
559
|
+
const key = path[lastIndex];
|
|
560
|
+
|
|
561
|
+
if (Object.hasOwn(target, key)) {
|
|
562
|
+
if (this.#dottedObjects.has(target[key])) {
|
|
563
|
+
this.#fail(`Cannot set ${formatPath(path)}: it is already an object built by dotted keys`, keyStart);
|
|
564
|
+
}
|
|
565
|
+
|
|
566
|
+
this.#fail(`Duplicate key ${formatPath(path)}`, keyStart);
|
|
567
|
+
}
|
|
568
|
+
|
|
569
|
+
defineMember(target, key, value);
|
|
570
|
+
}
|
|
571
|
+
|
|
572
|
+
#parseObject(depth) {
|
|
573
|
+
if (depth > MAX_DEPTH) {
|
|
574
|
+
this.#fail(`The document is nested more than ${MAX_DEPTH} levels deep`);
|
|
575
|
+
}
|
|
576
|
+
|
|
577
|
+
const start = this.#index;
|
|
578
|
+
const object = {};
|
|
579
|
+
this.#index++;
|
|
580
|
+
this.#skipTrivia();
|
|
581
|
+
|
|
582
|
+
for (;;) {
|
|
583
|
+
const code = this.#code();
|
|
584
|
+
|
|
585
|
+
if (code === CLOSE_BRACE) {
|
|
586
|
+
this.#index++;
|
|
587
|
+
return object;
|
|
588
|
+
}
|
|
589
|
+
|
|
590
|
+
if (this.#isAtEnd()) {
|
|
591
|
+
this.#fail('Unterminated object: expected "}"', start);
|
|
592
|
+
}
|
|
593
|
+
|
|
594
|
+
this.#parseEntry(object, depth);
|
|
595
|
+
this.#skipTrivia();
|
|
596
|
+
|
|
597
|
+
if (this.#code() === COMMA) {
|
|
598
|
+
this.#index++;
|
|
599
|
+
this.#skipTrivia();
|
|
600
|
+
} else if (this.#code() !== CLOSE_BRACE) {
|
|
601
|
+
this.#fail(this.#isAtEnd() ? 'Unterminated object: expected "}"' : `Expected "," or "}" after an object member, but found ${this.#describeHere()}`, this.#isAtEnd() ? start : this.#index);
|
|
602
|
+
}
|
|
603
|
+
}
|
|
604
|
+
}
|
|
605
|
+
|
|
606
|
+
#parseArray(depth) {
|
|
607
|
+
if (depth > MAX_DEPTH) {
|
|
608
|
+
this.#fail(`The document is nested more than ${MAX_DEPTH} levels deep`);
|
|
609
|
+
}
|
|
610
|
+
|
|
611
|
+
const start = this.#index;
|
|
612
|
+
const array = [];
|
|
613
|
+
this.#index++;
|
|
614
|
+
this.#skipTrivia();
|
|
615
|
+
|
|
616
|
+
for (;;) {
|
|
617
|
+
const code = this.#code();
|
|
618
|
+
|
|
619
|
+
if (code === CLOSE_BRACKET) {
|
|
620
|
+
this.#index++;
|
|
621
|
+
return array;
|
|
622
|
+
}
|
|
623
|
+
|
|
624
|
+
if (this.#isAtEnd()) {
|
|
625
|
+
this.#fail('Unterminated array: expected "]"', start);
|
|
626
|
+
}
|
|
627
|
+
|
|
628
|
+
const isRemoved = this.#parseRemoval();
|
|
629
|
+
const value = this.#parseValue(depth + 1);
|
|
630
|
+
|
|
631
|
+
if (!isRemoved) {
|
|
632
|
+
array.push(value);
|
|
633
|
+
}
|
|
634
|
+
|
|
635
|
+
this.#skipTrivia();
|
|
636
|
+
|
|
637
|
+
if (this.#code() === COMMA) {
|
|
638
|
+
this.#index++;
|
|
639
|
+
this.#skipTrivia();
|
|
640
|
+
} else if (this.#code() !== CLOSE_BRACKET) {
|
|
641
|
+
this.#fail(this.#isAtEnd() ? 'Unterminated array: expected "]"' : `Expected "," or "]" after an array item, but found ${this.#describeHere()}`, this.#isAtEnd() ? start : this.#index);
|
|
642
|
+
}
|
|
643
|
+
}
|
|
644
|
+
}
|
|
645
|
+
|
|
646
|
+
#parseValue(depth) {
|
|
647
|
+
const code = this.#code();
|
|
648
|
+
|
|
649
|
+
if (code === OPEN_BRACE) {
|
|
650
|
+
return this.#parseObject(depth);
|
|
651
|
+
}
|
|
652
|
+
|
|
653
|
+
if (code === OPEN_BRACKET) {
|
|
654
|
+
return this.#parseArray(depth);
|
|
655
|
+
}
|
|
656
|
+
|
|
657
|
+
let value;
|
|
658
|
+
|
|
659
|
+
if (code === SINGLE_QUOTE || code === DOUBLE_QUOTE) {
|
|
660
|
+
value = this.#parseString();
|
|
661
|
+
} else if (code === 0x74 /* t */ && this.#isKeyword('true')) {
|
|
662
|
+
value = true;
|
|
663
|
+
} else if (code === 0x66 /* f */ && this.#isKeyword('false')) {
|
|
664
|
+
value = false;
|
|
665
|
+
} else if (code === 0x6E /* n */ && this.#isKeyword('null')) {
|
|
666
|
+
value = null;
|
|
667
|
+
} else if (code === 0x69 /* i */ && this.#isKeyword('infinity')) {
|
|
668
|
+
value = Infinity;
|
|
669
|
+
} else if (code === DASH && this.#isKeyword('-infinity')) {
|
|
670
|
+
value = Number.NEGATIVE_INFINITY;
|
|
671
|
+
} else if ((code >= 0x30 && code <= 0x39) || code === DASH) {
|
|
672
|
+
value = this.#parseNumberOrInstant();
|
|
673
|
+
} else {
|
|
674
|
+
this.#failUnexpectedValue();
|
|
675
|
+
}
|
|
676
|
+
|
|
677
|
+
const next = this.#code();
|
|
678
|
+
|
|
679
|
+
if (!this.#isAtEnd() && (next >= 128 || valueEndCharacters[next] !== 1)) {
|
|
680
|
+
this.#fail(`Unexpected ${this.#describeHere()} after a value`);
|
|
681
|
+
}
|
|
682
|
+
|
|
683
|
+
return value;
|
|
684
|
+
}
|
|
685
|
+
|
|
686
|
+
#isKeyword(word) {
|
|
687
|
+
if (!this.#source.startsWith(word, this.#index)) {
|
|
688
|
+
return false;
|
|
689
|
+
}
|
|
690
|
+
|
|
691
|
+
const next = this.#code(this.#index + word.length);
|
|
692
|
+
|
|
693
|
+
// A longer word, such as `nullable`, is not the keyword.
|
|
694
|
+
if (next < 128 && bareKeyCharacters[next] === 1) {
|
|
695
|
+
return false;
|
|
696
|
+
}
|
|
697
|
+
|
|
698
|
+
this.#index += word.length;
|
|
699
|
+
return true;
|
|
700
|
+
}
|
|
701
|
+
|
|
702
|
+
#failUnexpectedValue() {
|
|
703
|
+
if (this.#isAtEnd()) {
|
|
704
|
+
this.#fail('Expected a value, but reached the end of the document');
|
|
705
|
+
}
|
|
706
|
+
|
|
707
|
+
const code = this.#code();
|
|
708
|
+
|
|
709
|
+
if (code === 0x2B /* + */) {
|
|
710
|
+
this.#fail('A "+" sign is not allowed. A number without a sign is positive');
|
|
711
|
+
}
|
|
712
|
+
|
|
713
|
+
if (code === DOT) {
|
|
714
|
+
this.#fail('A number cannot begin with "."; write a digit before it, as in 0.5');
|
|
715
|
+
}
|
|
716
|
+
|
|
717
|
+
let wordEnd = this.#index;
|
|
718
|
+
|
|
719
|
+
while (wordEnd < this.#source.length && this.#code(wordEnd) < 128 && bareKeyCharacters[this.#code(wordEnd)] === 1) {
|
|
720
|
+
wordEnd++;
|
|
721
|
+
}
|
|
722
|
+
|
|
723
|
+
if (wordEnd > this.#index) {
|
|
724
|
+
const word = this.#source.slice(this.#index, wordEnd);
|
|
725
|
+
|
|
726
|
+
// An entry whose value was left out, as in `a:` followed by `b: 1` on the next line, reads the next key as the value.
|
|
727
|
+
if (this.#code(wordEnd) === COLON) {
|
|
728
|
+
this.#fail(`Expected a value, but found the key ${abbreviate(word)}`);
|
|
729
|
+
}
|
|
730
|
+
|
|
731
|
+
this.#fail(describeUnknownWord(word));
|
|
732
|
+
}
|
|
733
|
+
|
|
734
|
+
this.#fail(`Expected a value, but found ${this.#describeHere()}`);
|
|
735
|
+
}
|
|
736
|
+
|
|
737
|
+
#describeHere() {
|
|
738
|
+
if (this.#isAtEnd()) {
|
|
739
|
+
return 'the end of the document';
|
|
740
|
+
}
|
|
741
|
+
|
|
742
|
+
const code = this.#source.codePointAt(this.#index);
|
|
743
|
+
|
|
744
|
+
return code === LF ? 'a line break' : describeCharacter(code);
|
|
745
|
+
}
|
|
746
|
+
|
|
747
|
+
#parseString() {
|
|
748
|
+
const source = this.#source;
|
|
749
|
+
const code = this.#code();
|
|
750
|
+
|
|
751
|
+
if (source.charCodeAt(this.#index + 1) === code && source.charCodeAt(this.#index + 2) === code) {
|
|
752
|
+
return this.#parseBlockString(code);
|
|
753
|
+
}
|
|
754
|
+
|
|
755
|
+
return code === SINGLE_QUOTE ? this.#parseLiteralString() : this.#parseEscapedString();
|
|
756
|
+
}
|
|
757
|
+
|
|
758
|
+
/*
|
|
759
|
+
`'...'` has no escapes, so its value is the source text between the quotes.
|
|
760
|
+
*/
|
|
761
|
+
#parseLiteralString() {
|
|
762
|
+
const source = this.#source;
|
|
763
|
+
const start = this.#index;
|
|
764
|
+
// Stops at the first quote or line break, so that a one-line document with many strings stays linear.
|
|
765
|
+
LITERAL_STRING_END.lastIndex = start + 1;
|
|
766
|
+
const match = LITERAL_STRING_END.exec(source);
|
|
767
|
+
const end = match?.index;
|
|
768
|
+
|
|
769
|
+
if (match === null || source.charCodeAt(end) === LF) {
|
|
770
|
+
this.#fail('Unterminated string. A \'...\' string must end on the line it starts on; use a block string (\'\'\') for multiple lines', start);
|
|
771
|
+
}
|
|
772
|
+
|
|
773
|
+
this.#index = end + 1;
|
|
774
|
+
return source.slice(start + 1, end);
|
|
775
|
+
}
|
|
776
|
+
|
|
777
|
+
#parseEscapedString() {
|
|
778
|
+
const source = this.#source;
|
|
779
|
+
const start = this.#index;
|
|
780
|
+
let chunkStart = start + 1;
|
|
781
|
+
let value = '';
|
|
782
|
+
|
|
783
|
+
for (;;) {
|
|
784
|
+
ESCAPED_STRING_SPECIAL.lastIndex = chunkStart;
|
|
785
|
+
const match = ESCAPED_STRING_SPECIAL.exec(source);
|
|
786
|
+
|
|
787
|
+
if (match === null || match[0] === '\n') {
|
|
788
|
+
this.#fail('Unterminated string. A "..." string must end on the line it starts on; use a block string (""") for multiple lines', start);
|
|
789
|
+
}
|
|
790
|
+
|
|
791
|
+
value += source.slice(chunkStart, match.index);
|
|
792
|
+
|
|
793
|
+
if (match[0] === '"') {
|
|
794
|
+
this.#index = match.index + 1;
|
|
795
|
+
return value;
|
|
796
|
+
}
|
|
797
|
+
|
|
798
|
+
const {text, end} = this.#parseEscape(match.index);
|
|
799
|
+
value += text;
|
|
800
|
+
chunkStart = end;
|
|
801
|
+
}
|
|
802
|
+
}
|
|
803
|
+
|
|
804
|
+
/*
|
|
805
|
+
Decodes the escape at `offset`, which is a backslash.
|
|
806
|
+
*/
|
|
807
|
+
#parseEscape(offset) {
|
|
808
|
+
const source = this.#source;
|
|
809
|
+
const character = source[offset + 1];
|
|
810
|
+
if (Object.hasOwn(SIMPLE_ESCAPES, character)) {
|
|
811
|
+
return {text: SIMPLE_ESCAPES[character], end: offset + 2};
|
|
812
|
+
}
|
|
813
|
+
|
|
814
|
+
if (character === 'u') {
|
|
815
|
+
UNICODE_ESCAPE.lastIndex = offset + 1;
|
|
816
|
+
const match = UNICODE_ESCAPE.exec(source);
|
|
817
|
+
|
|
818
|
+
if (match === null) {
|
|
819
|
+
this.#fail(describeBadUnicodeEscape(source, offset + 1), offset);
|
|
820
|
+
}
|
|
821
|
+
|
|
822
|
+
const {hex} = match.groups;
|
|
823
|
+
|
|
824
|
+
if (hex.length > 1 && hex.startsWith('0')) {
|
|
825
|
+
this.#fail(String.raw`A Unicode escape may not have leading zeros; write \u{${hex.replace(/^0+(?=.)/v, '')}}`, offset);
|
|
826
|
+
}
|
|
827
|
+
|
|
828
|
+
const codePoint = Number.parseInt(hex, 16);
|
|
829
|
+
|
|
830
|
+
if (codePoint === 0x0D) {
|
|
831
|
+
this.#fail(String.raw`A carriage return (U+000D) cannot be represented, so \u{d} is not allowed`, offset);
|
|
832
|
+
}
|
|
833
|
+
|
|
834
|
+
if (codePoint >= 0xD8_00 && codePoint <= 0xDF_FF) {
|
|
835
|
+
this.#fail(String.raw`\u{${hex}} is a surrogate, which is not a Unicode scalar value`, offset);
|
|
836
|
+
}
|
|
837
|
+
|
|
838
|
+
if (codePoint > 0x10_FF_FF) {
|
|
839
|
+
this.#fail(String.raw`\u{${hex}} is above U+10FFFF, the largest Unicode scalar value`, offset);
|
|
840
|
+
}
|
|
841
|
+
|
|
842
|
+
return {text: String.fromCodePoint(codePoint), end: offset + 1 + match[0].length};
|
|
843
|
+
}
|
|
844
|
+
|
|
845
|
+
if (character === 'r') {
|
|
846
|
+
this.#fail(String.raw`There is no \r escape, because a carriage return cannot be represented`, offset);
|
|
847
|
+
}
|
|
848
|
+
|
|
849
|
+
if (character === '\'') {
|
|
850
|
+
this.#fail('A \' needs no escape inside "..."', offset);
|
|
851
|
+
}
|
|
852
|
+
|
|
853
|
+
if (character === undefined || character === '\n') {
|
|
854
|
+
this.#fail(String.raw`A backslash must be followed by an escape character. Use \\ for a literal backslash, or a '...' string`, offset);
|
|
855
|
+
}
|
|
856
|
+
|
|
857
|
+
this.#fail(`Unknown escape "\\${String.fromCodePoint(source.codePointAt(offset + 1))}". The escapes are \\\\, \\", \\n, \\t, and \\u{…}; use a '...' string for literal backslashes`, offset);
|
|
858
|
+
}
|
|
859
|
+
|
|
860
|
+
/*
|
|
861
|
+
A block string opens with a run of three or more quotes and a line break, and closes at the first line whose first non-whitespace content is a run of exactly that many quotes. The closing line's indentation is removed from every content line, except a blank one, which holds only spaces and tabs and becomes an empty line.
|
|
862
|
+
*/
|
|
863
|
+
#parseBlockString(quote) {
|
|
864
|
+
const source = this.#source;
|
|
865
|
+
const start = this.#index;
|
|
866
|
+
let delimiterLength = 0;
|
|
867
|
+
|
|
868
|
+
while (source.charCodeAt(start + delimiterLength) === quote) {
|
|
869
|
+
delimiterLength++;
|
|
870
|
+
}
|
|
871
|
+
|
|
872
|
+
this.#index = start + delimiterLength;
|
|
873
|
+
|
|
874
|
+
if (this.#isAtEnd()) {
|
|
875
|
+
this.#fail('Unterminated block string', start);
|
|
876
|
+
}
|
|
877
|
+
|
|
878
|
+
// As in Swift, the opening delimiter is followed directly by a line break, not even by trailing whitespace.
|
|
879
|
+
if (this.#code() !== LF) {
|
|
880
|
+
this.#fail('A block string\'s opening delimiter must be followed directly by a line break, and its content starts on the next line');
|
|
881
|
+
}
|
|
882
|
+
|
|
883
|
+
this.#index++;
|
|
884
|
+
|
|
885
|
+
// Each content line as a pair of source offsets.
|
|
886
|
+
const lines = [];
|
|
887
|
+
let indentation;
|
|
888
|
+
|
|
889
|
+
for (;;) {
|
|
890
|
+
// Past the last line, either because the source ended with a line break or because the last line was content.
|
|
891
|
+
if (this.#index >= source.length) {
|
|
892
|
+
this.#fail('Unterminated block string', start);
|
|
893
|
+
}
|
|
894
|
+
|
|
895
|
+
const lineStart = this.#index;
|
|
896
|
+
let lineEnd = source.indexOf('\n', lineStart);
|
|
897
|
+
|
|
898
|
+
if (lineEnd === -1) {
|
|
899
|
+
lineEnd = source.length;
|
|
900
|
+
}
|
|
901
|
+
|
|
902
|
+
let contentStart = lineStart;
|
|
903
|
+
|
|
904
|
+
while (contentStart < lineEnd && (source.charCodeAt(contentStart) === SPACE || source.charCodeAt(contentStart) === TAB)) {
|
|
905
|
+
contentStart++;
|
|
906
|
+
}
|
|
907
|
+
|
|
908
|
+
let runLength = 0;
|
|
909
|
+
|
|
910
|
+
while (source.charCodeAt(contentStart + runLength) === quote) {
|
|
911
|
+
runLength++;
|
|
912
|
+
}
|
|
913
|
+
|
|
914
|
+
if (runLength === delimiterLength) {
|
|
915
|
+
indentation = source.slice(lineStart, contentStart);
|
|
916
|
+
this.#index = contentStart + delimiterLength;
|
|
917
|
+
break;
|
|
918
|
+
}
|
|
919
|
+
|
|
920
|
+
lines.push(lineStart, lineEnd);
|
|
921
|
+
this.#index = lineEnd + 1;
|
|
922
|
+
}
|
|
923
|
+
|
|
924
|
+
const contents = [];
|
|
925
|
+
|
|
926
|
+
for (let index = 0; index < lines.length; index += 2) {
|
|
927
|
+
const lineStart = lines[index];
|
|
928
|
+
const line = source.slice(lineStart, lines[index + 1]);
|
|
929
|
+
|
|
930
|
+
// A blank line, which is empty or holds only spaces and tabs, may leave out the indentation, and it becomes an empty line. Swift is stricter here: there only a completely empty line may leave it out.
|
|
931
|
+
if (BLANK_LINE.test(line)) {
|
|
932
|
+
contents.push({text: '', offset: lineStart});
|
|
933
|
+
} else if (line.startsWith(indentation)) {
|
|
934
|
+
contents.push({text: line.slice(indentation.length), offset: lineStart + indentation.length});
|
|
935
|
+
} else {
|
|
936
|
+
this.#fail('This line does not start with the indentation of its block string\'s closing delimiter. Every line except a blank one must start with exactly the same spaces and tabs', lineStart);
|
|
937
|
+
}
|
|
938
|
+
}
|
|
939
|
+
|
|
940
|
+
// Blank lines directly after the opening delimiter and directly before the closing one are not content. Every blank line is empty by now.
|
|
941
|
+
let first = 0;
|
|
942
|
+
let last = contents.length;
|
|
943
|
+
|
|
944
|
+
while (first < last && contents[first].text === '') {
|
|
945
|
+
first++;
|
|
946
|
+
}
|
|
947
|
+
|
|
948
|
+
while (last > first && contents[last - 1].text === '') {
|
|
949
|
+
last--;
|
|
950
|
+
}
|
|
951
|
+
|
|
952
|
+
const kept = contents.slice(first, last);
|
|
953
|
+
|
|
954
|
+
return quote === SINGLE_QUOTE ? kept.map(line => line.text).join('\n') : kept.map(line => this.#unescapeLine(line.text, line.offset)).join('\n');
|
|
955
|
+
}
|
|
956
|
+
|
|
957
|
+
#unescapeLine(text, offset) {
|
|
958
|
+
let value = '';
|
|
959
|
+
let chunkStart = 0;
|
|
960
|
+
|
|
961
|
+
for (let index = text.indexOf('\\'); index !== -1; index = text.indexOf('\\', chunkStart)) {
|
|
962
|
+
value += text.slice(chunkStart, index);
|
|
963
|
+
const escape = this.#parseEscape(offset + index);
|
|
964
|
+
value += escape.text;
|
|
965
|
+
chunkStart = escape.end - offset;
|
|
966
|
+
}
|
|
967
|
+
|
|
968
|
+
return value + text.slice(chunkStart);
|
|
969
|
+
}
|
|
970
|
+
|
|
971
|
+
#parseNumberOrInstant() {
|
|
972
|
+
const plain = this.#parsePlainNumber();
|
|
973
|
+
|
|
974
|
+
if (plain !== undefined) {
|
|
975
|
+
return plain;
|
|
976
|
+
}
|
|
977
|
+
|
|
978
|
+
const source = this.#source;
|
|
979
|
+
const start = this.#index;
|
|
980
|
+
let end = start + 1;
|
|
981
|
+
|
|
982
|
+
while (end < source.length && source.charCodeAt(end) < 128 && numberCharacters[source.charCodeAt(end)] === 1) {
|
|
983
|
+
end++;
|
|
984
|
+
}
|
|
985
|
+
|
|
986
|
+
const text = source.slice(start, end);
|
|
987
|
+
this.#index = end;
|
|
988
|
+
|
|
989
|
+
if (INSTANT_PREFIX.test(text)) {
|
|
990
|
+
return this.#parseInstant(text, start);
|
|
991
|
+
}
|
|
992
|
+
|
|
993
|
+
const kind = classifyNumber(text);
|
|
994
|
+
|
|
995
|
+
if (kind === undefined && isDurationLike(text)) {
|
|
996
|
+
return this.#parseDuration(text, start);
|
|
997
|
+
}
|
|
998
|
+
|
|
999
|
+
switch (kind) {
|
|
1000
|
+
case 'decimal':
|
|
1001
|
+
case 'radix': {
|
|
1002
|
+
if (text === '-0') {
|
|
1003
|
+
this.#fail('"-0" is not allowed, because zero has one spelling: 0', start);
|
|
1004
|
+
}
|
|
1005
|
+
|
|
1006
|
+
return this.#integer(text.replaceAll('_', ''), text, start);
|
|
1007
|
+
}
|
|
1008
|
+
|
|
1009
|
+
case 'float': {
|
|
1010
|
+
const value = Number(text.replaceAll('_', ''));
|
|
1011
|
+
|
|
1012
|
+
if (!Number.isFinite(value)) {
|
|
1013
|
+
this.#fail(`${abbreviate(text)} is too large to be a finite float. Use infinity if you mean it`, start);
|
|
1014
|
+
}
|
|
1015
|
+
|
|
1016
|
+
// Negative zero is the same value as zero.
|
|
1017
|
+
return value === 0 ? 0 : value;
|
|
1018
|
+
}
|
|
1019
|
+
|
|
1020
|
+
default: {
|
|
1021
|
+
break;
|
|
1022
|
+
}
|
|
1023
|
+
}
|
|
1024
|
+
|
|
1025
|
+
this.#fail(describeBadNumber(text), start);
|
|
1026
|
+
}
|
|
1027
|
+
|
|
1028
|
+
/*
|
|
1029
|
+
The common case, a short decimal int or float such as `8080`, `-3`, or `30.5`, without the general path's maximal-run scan and classification. Anything else, including every error, returns `undefined` and is left to the general path.
|
|
1030
|
+
*/
|
|
1031
|
+
#parsePlainNumber() {
|
|
1032
|
+
const source = this.#source;
|
|
1033
|
+
const start = this.#index;
|
|
1034
|
+
let index = source.charCodeAt(start) === DASH ? start + 1 : start;
|
|
1035
|
+
const integerStart = index;
|
|
1036
|
+
|
|
1037
|
+
while (isDigit(source.charCodeAt(index))) {
|
|
1038
|
+
index++;
|
|
1039
|
+
}
|
|
1040
|
+
|
|
1041
|
+
const integerLength = index - integerStart;
|
|
1042
|
+
|
|
1043
|
+
// No digits, or a leading zero, which is either `0` alone or an error.
|
|
1044
|
+
if (integerLength === 0 || (integerLength > 1 && source.charCodeAt(integerStart) === 0x30)) {
|
|
1045
|
+
return;
|
|
1046
|
+
}
|
|
1047
|
+
|
|
1048
|
+
let isFloat = false;
|
|
1049
|
+
|
|
1050
|
+
if (source.charCodeAt(index) === DOT) {
|
|
1051
|
+
const fractionStart = index + 1;
|
|
1052
|
+
index = fractionStart;
|
|
1053
|
+
|
|
1054
|
+
while (isDigit(source.charCodeAt(index))) {
|
|
1055
|
+
index++;
|
|
1056
|
+
}
|
|
1057
|
+
|
|
1058
|
+
if (index === fractionStart) {
|
|
1059
|
+
return;
|
|
1060
|
+
}
|
|
1061
|
+
|
|
1062
|
+
isFloat = true;
|
|
1063
|
+
}
|
|
1064
|
+
|
|
1065
|
+
const next = source.charCodeAt(index);
|
|
1066
|
+
|
|
1067
|
+
// At most 15 digits, so an int is exact as a number. A character that may continue a number, such as `e`, `_`, or `-`, needs the general path.
|
|
1068
|
+
if (index - integerStart > 15 || (index < source.length && (next >= 128 || valueEndCharacters[next] !== 1))) {
|
|
1069
|
+
return;
|
|
1070
|
+
}
|
|
1071
|
+
|
|
1072
|
+
const value = Number(source.slice(start, index));
|
|
1073
|
+
|
|
1074
|
+
if (isFloat) {
|
|
1075
|
+
this.#index = index;
|
|
1076
|
+
|
|
1077
|
+
// Negative zero is the same value as zero.
|
|
1078
|
+
return value === 0 ? 0 : value;
|
|
1079
|
+
}
|
|
1080
|
+
|
|
1081
|
+
// `-0` is an error, which the general path reports.
|
|
1082
|
+
if (value === 0 && index - start > 1) {
|
|
1083
|
+
return;
|
|
1084
|
+
}
|
|
1085
|
+
|
|
1086
|
+
this.#index = index;
|
|
1087
|
+
return this.#integers === 'bigint' ? BigInt(value) : value;
|
|
1088
|
+
}
|
|
1089
|
+
|
|
1090
|
+
/*
|
|
1091
|
+
`digits` is the int without underscores. More than 64 significant digits is out of range in any radix, which is decided before `BigInt` spends time on a huge digit string.
|
|
1092
|
+
*/
|
|
1093
|
+
#integer(digits, text, start) {
|
|
1094
|
+
const isRadix = digits.length > 1 && digits.charCodeAt(1) > 0x39;
|
|
1095
|
+
let significant = digits.charCodeAt(0) === DASH ? 1 : 0;
|
|
1096
|
+
|
|
1097
|
+
if (isRadix) {
|
|
1098
|
+
significant = 2;
|
|
1099
|
+
|
|
1100
|
+
while (digits.charCodeAt(significant) === 0x30) {
|
|
1101
|
+
significant++;
|
|
1102
|
+
}
|
|
1103
|
+
}
|
|
1104
|
+
|
|
1105
|
+
const value = digits.length - significant > 64 ? undefined : BigInt(isRadix ? `${digits.slice(0, 2)}${digits.slice(significant) || '0'}` : digits);
|
|
1106
|
+
|
|
1107
|
+
if (value === undefined || value < INT64_MIN || value > INT64_MAX) {
|
|
1108
|
+
this.#fail(`The integer ${abbreviate(text)} is outside the 64-bit range (-9223372036854775808 to 9223372036854775807)`, start);
|
|
1109
|
+
}
|
|
1110
|
+
|
|
1111
|
+
if (this.#integers === 'bigint') {
|
|
1112
|
+
return value;
|
|
1113
|
+
}
|
|
1114
|
+
|
|
1115
|
+
if (value < Number.MIN_SAFE_INTEGER || value > Number.MAX_SAFE_INTEGER) {
|
|
1116
|
+
this.#fail(`The integer ${abbreviate(text)} cannot be represented exactly as a JavaScript number. Remove the \`integers: 'number'\` option to get a BigInt`, start);
|
|
1117
|
+
}
|
|
1118
|
+
|
|
1119
|
+
return Number(value);
|
|
1120
|
+
}
|
|
1121
|
+
|
|
1122
|
+
#parseInstant(text, start) {
|
|
1123
|
+
const match = text.length > MAX_DIAGNOSED_LENGTH ? null : INSTANT.exec(text);
|
|
1124
|
+
|
|
1125
|
+
if (match === null) {
|
|
1126
|
+
this.#fail(describeBadInstant(text), start);
|
|
1127
|
+
}
|
|
1128
|
+
|
|
1129
|
+
const {year, month, day, hour, minute, second, fraction, offset, offsetHour, offsetMinute} = match.groups;
|
|
1130
|
+
const check = (isValid, reason) => {
|
|
1131
|
+
if (!isValid) {
|
|
1132
|
+
this.#fail(`Invalid instant ${abbreviate(text)}: ${reason}`, start);
|
|
1133
|
+
}
|
|
1134
|
+
};
|
|
1135
|
+
|
|
1136
|
+
check(fraction === undefined || fraction.length <= 9, 'a fractional second has at most nine digits');
|
|
1137
|
+
check(year !== '0000', 'the year must be 0001 to 9999');
|
|
1138
|
+
check(month >= '01' && month <= '12', 'the month must be 01 to 12');
|
|
1139
|
+
const lastDay = daysInMonth(Number(year), Number(month));
|
|
1140
|
+
check(day >= '01' && Number(day) <= lastDay, `the day must be 01 to ${lastDay} in that month`);
|
|
1141
|
+
check(hour <= '23', 'the hour must be 00 to 23');
|
|
1142
|
+
check(minute <= '59', 'the minute must be 00 to 59');
|
|
1143
|
+
check(second <= '59', 'the second must be 00 to 59, and a leap second is not representable');
|
|
1144
|
+
|
|
1145
|
+
if (offset !== 'Z') {
|
|
1146
|
+
check(offsetHour <= '23', 'the offset hour must be 00 to 23');
|
|
1147
|
+
check(offsetMinute <= '59', 'the offset minute must be 00 to 59');
|
|
1148
|
+
check(offset !== '-00:00', '-00:00 means "offset unknown" in RFC 3339, which is not representable; use Z or +00:00');
|
|
1149
|
+
}
|
|
1150
|
+
|
|
1151
|
+
const instant = Temporal.Instant.from(text);
|
|
1152
|
+
const nanoseconds = instant.epochNanoseconds;
|
|
1153
|
+
check(nanoseconds >= MIN_INSTANT && nanoseconds <= MAX_INSTANT, 'in UTC it falls outside the years 0001 to 9999');
|
|
1154
|
+
return instant;
|
|
1155
|
+
}
|
|
1156
|
+
|
|
1157
|
+
/*
|
|
1158
|
+
A scan of the parts in order, so that each error names the part it is about.
|
|
1159
|
+
*/
|
|
1160
|
+
#parseDuration(text, start) {
|
|
1161
|
+
const fail = reason => {
|
|
1162
|
+
this.#fail(`Invalid duration ${abbreviate(text)}: ${reason}`, start);
|
|
1163
|
+
};
|
|
1164
|
+
|
|
1165
|
+
const isNegative = text.charCodeAt(0) === DASH;
|
|
1166
|
+
let index = isNegative ? 1 : 0;
|
|
1167
|
+
let previousRank = -1;
|
|
1168
|
+
let total = 0n;
|
|
1169
|
+
// The fraction of the last part, and that part's unit size. Only the last part may have one, so the value checks wait until every part is read.
|
|
1170
|
+
let fraction = '';
|
|
1171
|
+
let size = 0n;
|
|
1172
|
+
|
|
1173
|
+
while (index < text.length) {
|
|
1174
|
+
const digitsEnd = skipIntegerPart(text, index);
|
|
1175
|
+
const next = text.charCodeAt(digitsEnd);
|
|
1176
|
+
|
|
1177
|
+
if (digitsEnd === index) {
|
|
1178
|
+
const code = text.charCodeAt(index);
|
|
1179
|
+
|
|
1180
|
+
if (code === DASH && isDigit(text.charCodeAt(index + 1))) {
|
|
1181
|
+
fail('only the whole duration takes a sign, as in -1h30m');
|
|
1182
|
+
}
|
|
1183
|
+
|
|
1184
|
+
fail(code === 0x2B /* + */ ? 'a "+" sign is not allowed' : `expected a number at "${abbreviate(text.slice(index), 10)}"`);
|
|
1185
|
+
}
|
|
1186
|
+
|
|
1187
|
+
if (text.charCodeAt(index) === 0x30 && (isDigit(next) || next === 0x5F /* _ */)) {
|
|
1188
|
+
fail('leading zeros are not allowed');
|
|
1189
|
+
}
|
|
1190
|
+
|
|
1191
|
+
if (next === 0x5F /* _ */) {
|
|
1192
|
+
fail('an underscore must be between two digits');
|
|
1193
|
+
}
|
|
1194
|
+
|
|
1195
|
+
fraction = '';
|
|
1196
|
+
let unitStart = digitsEnd;
|
|
1197
|
+
|
|
1198
|
+
if (next === DOT) {
|
|
1199
|
+
unitStart = skipDigits(text, digitsEnd + 1, isDigit);
|
|
1200
|
+
|
|
1201
|
+
if (unitStart === digitsEnd + 1) {
|
|
1202
|
+
fail('a "." must be followed by a digit');
|
|
1203
|
+
}
|
|
1204
|
+
|
|
1205
|
+
if (text.charCodeAt(unitStart) === 0x5F /* _ */) {
|
|
1206
|
+
fail('an underscore must be between two digits');
|
|
1207
|
+
}
|
|
1208
|
+
|
|
1209
|
+
// Trailing zeros change nothing, and leaving them out keeps the arithmetic small however many there are.
|
|
1210
|
+
fraction = trimTrailingZeros(text.slice(digitsEnd + 1, unitStart).replaceAll('_', ''));
|
|
1211
|
+
}
|
|
1212
|
+
|
|
1213
|
+
let unitEnd = unitStart;
|
|
1214
|
+
|
|
1215
|
+
while (isLetter(text.charCodeAt(unitEnd))) {
|
|
1216
|
+
unitEnd++;
|
|
1217
|
+
}
|
|
1218
|
+
|
|
1219
|
+
const unit = text.slice(unitStart, unitEnd);
|
|
1220
|
+
const rank = DURATION_UNIT_NAMES.indexOf(unit);
|
|
1221
|
+
|
|
1222
|
+
if (rank === -1) {
|
|
1223
|
+
fail(describeBadDurationUnit(unit));
|
|
1224
|
+
}
|
|
1225
|
+
|
|
1226
|
+
if (rank <= previousRank) {
|
|
1227
|
+
fail('the units are in the order h, m, s, ms, us, ns, and each appears at most once');
|
|
1228
|
+
}
|
|
1229
|
+
|
|
1230
|
+
if (next === DOT && isDigit(text.charCodeAt(unitEnd))) {
|
|
1231
|
+
fail('only the last part may have a fraction');
|
|
1232
|
+
}
|
|
1233
|
+
|
|
1234
|
+
size = DURATION_UNITS.get(unit);
|
|
1235
|
+
const digits = text.slice(index, digitsEnd).replaceAll('_', '');
|
|
1236
|
+
// More than 19 digits is out of range in any unit, which is decided before `BigInt` spends time on a huge digit string.
|
|
1237
|
+
total += digits.length > 19 ? INT64_MAX * 2n : BigInt(digits) * size;
|
|
1238
|
+
index = unitEnd;
|
|
1239
|
+
previousRank = rank;
|
|
1240
|
+
}
|
|
1241
|
+
|
|
1242
|
+
// More than 13 fraction digits without a trailing zero is finer than a nanosecond in any unit, which is decided before `BigInt` spends time on a huge digit string.
|
|
1243
|
+
if (fraction.length > 13) {
|
|
1244
|
+
fail('it is not a whole number of nanoseconds');
|
|
1245
|
+
}
|
|
1246
|
+
|
|
1247
|
+
if (fraction !== '') {
|
|
1248
|
+
const scale = 10n ** BigInt(fraction.length);
|
|
1249
|
+
const fractionNanoseconds = BigInt(fraction) * size;
|
|
1250
|
+
|
|
1251
|
+
if (fractionNanoseconds % scale !== 0n) {
|
|
1252
|
+
fail('it is not a whole number of nanoseconds');
|
|
1253
|
+
}
|
|
1254
|
+
|
|
1255
|
+
total += fractionNanoseconds / scale;
|
|
1256
|
+
}
|
|
1257
|
+
|
|
1258
|
+
if (isNegative && total === 0n) {
|
|
1259
|
+
fail('"-" is not allowed before zero, because zero has one spelling: 0s');
|
|
1260
|
+
}
|
|
1261
|
+
|
|
1262
|
+
if (total > (isNegative ? -INT64_MIN : INT64_MAX)) {
|
|
1263
|
+
fail('it is outside the 64-bit range of nanoseconds, about 292 years either way');
|
|
1264
|
+
}
|
|
1265
|
+
|
|
1266
|
+
return createDuration(isNegative ? -total : total);
|
|
1267
|
+
}
|
|
1268
|
+
}
|
|
1269
|
+
|
|
1270
|
+
/*
|
|
1271
|
+
Whether `text` is a decimal int, a radix int, or a float, following the grammar. A run of digits may have single underscores between digits.
|
|
1272
|
+
*/
|
|
1273
|
+
function classifyNumber(text) {
|
|
1274
|
+
const radix = text.charCodeAt(0) === 0x30 ? RADIX_DIGIT[text[1]] : undefined;
|
|
1275
|
+
|
|
1276
|
+
if (radix !== undefined) {
|
|
1277
|
+
const end = skipDigits(text, 2, radix);
|
|
1278
|
+
return end > 2 && end === text.length ? 'radix' : undefined;
|
|
1279
|
+
}
|
|
1280
|
+
|
|
1281
|
+
let index = text.charCodeAt(0) === DASH ? 1 : 0;
|
|
1282
|
+
|
|
1283
|
+
// The integer part is `0` or has no leading zero.
|
|
1284
|
+
const integerEnd = skipIntegerPart(text, index);
|
|
1285
|
+
|
|
1286
|
+
if (integerEnd === index) {
|
|
1287
|
+
return undefined;
|
|
1288
|
+
}
|
|
1289
|
+
|
|
1290
|
+
index = integerEnd;
|
|
1291
|
+
|
|
1292
|
+
let kind = 'decimal';
|
|
1293
|
+
|
|
1294
|
+
if (text.charCodeAt(index) === DOT) {
|
|
1295
|
+
const end = skipDigits(text, index + 1, isDigit);
|
|
1296
|
+
|
|
1297
|
+
if (end === index + 1) {
|
|
1298
|
+
return undefined;
|
|
1299
|
+
}
|
|
1300
|
+
|
|
1301
|
+
index = end;
|
|
1302
|
+
kind = 'float';
|
|
1303
|
+
}
|
|
1304
|
+
|
|
1305
|
+
if (text.charCodeAt(index) === 0x65 /* e */) {
|
|
1306
|
+
const isNegative = text.charCodeAt(index + 1) === DASH;
|
|
1307
|
+
const exponentStart = index + (isNegative ? 2 : 1);
|
|
1308
|
+
// The exponent follows the same leading-zero rule as the integer part, and its zero has one spelling, `e0`, so `e-0` is not an exponent.
|
|
1309
|
+
const end = isNegative && text.charCodeAt(exponentStart) === 0x30 ? exponentStart : skipIntegerPart(text, exponentStart);
|
|
1310
|
+
|
|
1311
|
+
if (end === exponentStart) {
|
|
1312
|
+
return undefined;
|
|
1313
|
+
}
|
|
1314
|
+
|
|
1315
|
+
index = end;
|
|
1316
|
+
kind = 'float';
|
|
1317
|
+
}
|
|
1318
|
+
|
|
1319
|
+
return index === text.length ? kind : undefined;
|
|
1320
|
+
}
|
|
1321
|
+
|
|
1322
|
+
/*
|
|
1323
|
+
Returns the index after a run of digits that may have single underscores between them, or `index` when there is no digit there.
|
|
1324
|
+
*/
|
|
1325
|
+
function skipDigits(text, index, isValidDigit) {
|
|
1326
|
+
if (!isValidDigit(text.charCodeAt(index))) {
|
|
1327
|
+
return index;
|
|
1328
|
+
}
|
|
1329
|
+
|
|
1330
|
+
index++;
|
|
1331
|
+
|
|
1332
|
+
for (;;) {
|
|
1333
|
+
const code = text.charCodeAt(index);
|
|
1334
|
+
|
|
1335
|
+
if (isValidDigit(code)) {
|
|
1336
|
+
index++;
|
|
1337
|
+
} else if (code === 0x5F /* _ */ && isValidDigit(text.charCodeAt(index + 1))) {
|
|
1338
|
+
index += 2;
|
|
1339
|
+
} else {
|
|
1340
|
+
return index;
|
|
1341
|
+
}
|
|
1342
|
+
}
|
|
1343
|
+
}
|
|
1344
|
+
|
|
1345
|
+
/*
|
|
1346
|
+
Returns the index after an integer part, which is `0` or digits without a leading zero, or `index` when there is none. Anything after a leading `0`, such as the `5` in `05`, is left for the caller to reject.
|
|
1347
|
+
*/
|
|
1348
|
+
function skipIntegerPart(text, index) {
|
|
1349
|
+
return text.charCodeAt(index) === 0x30 ? index + 1 : skipDigits(text, index, isDigit);
|
|
1350
|
+
}
|
|
1351
|
+
|
|
1352
|
+
function isDigit(code) {
|
|
1353
|
+
return code >= 0x30 && code <= 0x39;
|
|
1354
|
+
}
|
|
1355
|
+
|
|
1356
|
+
function isLetter(code) {
|
|
1357
|
+
return (code >= 0x41 && code <= 0x5A) || (code >= 0x61 && code <= 0x7A);
|
|
1358
|
+
}
|
|
1359
|
+
|
|
1360
|
+
/*
|
|
1361
|
+
Whether a token that is not a number or an instant was meant as a duration: its first letter could begin a unit, or a day or a week. A radix prefix, an exponent, and other letters, such as the `T` in `20260919T140000Z` or the `x` in `1.5x`, are left to the number errors.
|
|
1362
|
+
*/
|
|
1363
|
+
function isDurationLike(text) {
|
|
1364
|
+
const start = text.charCodeAt(0) === DASH ? 1 : 0;
|
|
1365
|
+
|
|
1366
|
+
if (!isDigit(text.charCodeAt(start))) {
|
|
1367
|
+
return false;
|
|
1368
|
+
}
|
|
1369
|
+
|
|
1370
|
+
// A scan rather than a regular expression, because a fraction may have any number of trailing zeros, so the first letter can be millions of characters in.
|
|
1371
|
+
let index = start;
|
|
1372
|
+
|
|
1373
|
+
while (isDigit(text.charCodeAt(index)) || text.charCodeAt(index) === 0x5F) {
|
|
1374
|
+
index++;
|
|
1375
|
+
}
|
|
1376
|
+
|
|
1377
|
+
if (text.charCodeAt(index) === DOT) {
|
|
1378
|
+
index++;
|
|
1379
|
+
|
|
1380
|
+
while (isDigit(text.charCodeAt(index)) || text.charCodeAt(index) === 0x5F) {
|
|
1381
|
+
index++;
|
|
1382
|
+
}
|
|
1383
|
+
}
|
|
1384
|
+
|
|
1385
|
+
return /^[dhmnsuw]$/iv.test(text.charAt(index));
|
|
1386
|
+
}
|
|
1387
|
+
|
|
1388
|
+
function describeBadDurationUnit(unit) {
|
|
1389
|
+
if (unit === '') {
|
|
1390
|
+
return 'every number needs a unit: h, m, s, ms, us, or ns';
|
|
1391
|
+
}
|
|
1392
|
+
|
|
1393
|
+
if (/^d(?:ays?)?$/v.test(unit)) {
|
|
1394
|
+
return 'there is no day unit, because a day is not a fixed length. Write 24h for a fixed 24 hours';
|
|
1395
|
+
}
|
|
1396
|
+
|
|
1397
|
+
if (/^w(?:eeks?)?$/v.test(unit)) {
|
|
1398
|
+
return 'there is no week unit, because a day is not a fixed length. Write 168h for a fixed 168 hours';
|
|
1399
|
+
}
|
|
1400
|
+
|
|
1401
|
+
if (DURATION_UNITS.has(unit.toLowerCase())) {
|
|
1402
|
+
return `the units are lowercase: ${unit.toLowerCase()}`;
|
|
1403
|
+
}
|
|
1404
|
+
|
|
1405
|
+
return `"${abbreviate(unit, 10)}" is not a unit. The units are h, m, s, ms, us, and ns, and a string must be quoted`;
|
|
1406
|
+
}
|
|
1407
|
+
|
|
1408
|
+
function isWordsWithSpaces(text) {
|
|
1409
|
+
for (let index = 0; index < text.length; index++) {
|
|
1410
|
+
const code = text.charCodeAt(index);
|
|
1411
|
+
|
|
1412
|
+
if (code !== SPACE && code !== TAB && (code >= 128 || bareKeyCharacters[code] !== 1)) {
|
|
1413
|
+
return false;
|
|
1414
|
+
}
|
|
1415
|
+
}
|
|
1416
|
+
|
|
1417
|
+
return true;
|
|
1418
|
+
}
|
|
1419
|
+
|
|
1420
|
+
function findLineEnd(source, offset) {
|
|
1421
|
+
const end = source.indexOf('\n', offset);
|
|
1422
|
+
return end === -1 ? source.length : end;
|
|
1423
|
+
}
|
|
1424
|
+
|
|
1425
|
+
function daysInMonth(year, month) {
|
|
1426
|
+
if (month === 2) {
|
|
1427
|
+
const isLeapYear = year % 4 === 0 && (year % 100 !== 0 || year % 400 === 0);
|
|
1428
|
+
return isLeapYear ? 29 : 28;
|
|
1429
|
+
}
|
|
1430
|
+
|
|
1431
|
+
return [4, 6, 9, 11].includes(month) ? 30 : 31;
|
|
1432
|
+
}
|
|
1433
|
+
|
|
1434
|
+
function describeUnknownWord(word) {
|
|
1435
|
+
const lowercase = word.toLowerCase();
|
|
1436
|
+
|
|
1437
|
+
if (['true', 'false', 'yes', 'no', 'on', 'off'].includes(lowercase)) {
|
|
1438
|
+
return `"${word}" is not a value. Booleans are written true and false, in lowercase`;
|
|
1439
|
+
}
|
|
1440
|
+
|
|
1441
|
+
if (['null', 'nil', 'none', 'undefined'].includes(lowercase)) {
|
|
1442
|
+
return `"${word}" is not a value. Null is written null, in lowercase`;
|
|
1443
|
+
}
|
|
1444
|
+
|
|
1445
|
+
if (lowercase === 'nan') {
|
|
1446
|
+
return 'NaN is not representable. Use null for a missing value';
|
|
1447
|
+
}
|
|
1448
|
+
|
|
1449
|
+
return ['inf', 'infinity'].includes(lowercase) ? `"${word}" is not a value. Infinity is written infinity, in lowercase` : `Unexpected "${abbreviate(word)}". A string value must be quoted, as in '${abbreviate(word)}'`;
|
|
1450
|
+
}
|
|
1451
|
+
|
|
1452
|
+
function describeBadUnicodeEscape(source, offset) {
|
|
1453
|
+
const rest = source.slice(offset, offset + 12);
|
|
1454
|
+
|
|
1455
|
+
if (/^u[\dA-Fa-f]{4}/v.test(rest)) {
|
|
1456
|
+
return `The four-digit \\${rest.slice(0, 5)} form is not an escape. Write \\u{${rest.slice(1, 5).toLowerCase().replace(/^0+(?=.)/v, '')}}`;
|
|
1457
|
+
}
|
|
1458
|
+
|
|
1459
|
+
if (/^u\{[\dA-Fa-f]*[A-F]/v.test(rest)) {
|
|
1460
|
+
return 'A Unicode escape uses lowercase hexadecimal digits';
|
|
1461
|
+
}
|
|
1462
|
+
|
|
1463
|
+
if (/^u\{\}/v.test(rest)) {
|
|
1464
|
+
return 'A Unicode escape needs one to six hexadecimal digits';
|
|
1465
|
+
}
|
|
1466
|
+
|
|
1467
|
+
return /^u\{[\da-f]{7}/v.test(rest) ? 'A Unicode escape has at most six hexadecimal digits' : String.raw`A Unicode escape is written \u{…} with one to six lowercase hexadecimal digits`;
|
|
1468
|
+
}
|
|
1469
|
+
|
|
1470
|
+
function describeBadNumber(fullText) {
|
|
1471
|
+
const text = fullText.slice(0, MAX_DIAGNOSED_LENGTH);
|
|
1472
|
+
|
|
1473
|
+
if (text.includes('+')) {
|
|
1474
|
+
return 'A "+" sign is not allowed in a number, including in an exponent';
|
|
1475
|
+
}
|
|
1476
|
+
|
|
1477
|
+
if (/^-?0[BOX]/v.test(text)) {
|
|
1478
|
+
return 'A number prefix is lowercase: 0x, 0o, or 0b';
|
|
1479
|
+
}
|
|
1480
|
+
|
|
1481
|
+
const radixMatch = /^(?<sign>-?)0(?<radix>[box])(?<digits>.*)$/v.exec(text);
|
|
1482
|
+
|
|
1483
|
+
if (radixMatch !== null) {
|
|
1484
|
+
const {sign, radix, digits} = radixMatch.groups;
|
|
1485
|
+
const name = RADIX_NAME[radix];
|
|
1486
|
+
|
|
1487
|
+
if (sign === '-') {
|
|
1488
|
+
return `${RADIX_ARTICLE[radix]} ${name} integer cannot have a sign, because it states a bit pattern rather than a quantity`;
|
|
1489
|
+
}
|
|
1490
|
+
|
|
1491
|
+
if (digits === '') {
|
|
1492
|
+
return `Expected ${name} digits after "0${radix}"`;
|
|
1493
|
+
}
|
|
1494
|
+
|
|
1495
|
+
if (radix === 'x' && /[a-f]/v.test(digits) && !/[^\da-f_]/iv.test(digits)) {
|
|
1496
|
+
return `Hexadecimal digits are uppercase: 0x${abbreviate(digits.toUpperCase())}`;
|
|
1497
|
+
}
|
|
1498
|
+
|
|
1499
|
+
const validCharacter = {x: /[\dA-F_]/v, o: /[0-7_]/v, b: /[01_]/v}[radix];
|
|
1500
|
+
const character = [...digits].find(character => !validCharacter.test(character));
|
|
1501
|
+
|
|
1502
|
+
return character === undefined ? 'An underscore in a number must be between two digits' : `Invalid ${name} digit "${character}"`;
|
|
1503
|
+
}
|
|
1504
|
+
|
|
1505
|
+
if (/^-?\d[\d_]*E/v.test(text) || /^-?\d[\d_]*\.[\d_]+E/v.test(text)) {
|
|
1506
|
+
return 'An exponent marker is a lowercase "e"';
|
|
1507
|
+
}
|
|
1508
|
+
|
|
1509
|
+
if (/^-?[\d_]+\.(?:$|[^\d_])/v.test(text)) {
|
|
1510
|
+
return 'A decimal point must be followed by a digit';
|
|
1511
|
+
}
|
|
1512
|
+
|
|
1513
|
+
if (/^-\./v.test(text)) {
|
|
1514
|
+
return 'A number cannot begin with "."; write a digit before it, as in -0.5';
|
|
1515
|
+
}
|
|
1516
|
+
|
|
1517
|
+
if (/^-?0[\d_]/v.test(text)) {
|
|
1518
|
+
return 'Leading zeros are not allowed in a decimal number';
|
|
1519
|
+
}
|
|
1520
|
+
|
|
1521
|
+
if (/_(?:$|\D)|(?:^|\D)_/v.test(text)) {
|
|
1522
|
+
return 'An underscore in a number must be between two digits';
|
|
1523
|
+
}
|
|
1524
|
+
|
|
1525
|
+
if (/^-?\d[\d_]*(?:\.[\d_]+)?e-?0_?\d/v.test(text)) {
|
|
1526
|
+
return 'Leading zeros are not allowed in an exponent';
|
|
1527
|
+
}
|
|
1528
|
+
|
|
1529
|
+
if (/^-?\d[\d_]*(?:\.[\d_]+)?e-0$/v.test(text)) {
|
|
1530
|
+
return '"e-0" is not allowed, because an exponent of zero has one spelling: e0';
|
|
1531
|
+
}
|
|
1532
|
+
|
|
1533
|
+
if (/e-?$/v.test(text)) {
|
|
1534
|
+
return 'Expected digits after the exponent marker "e"';
|
|
1535
|
+
}
|
|
1536
|
+
|
|
1537
|
+
if (text.split('.').length > 2) {
|
|
1538
|
+
return `Invalid number "${abbreviate(text)}". A value with several dots, such as a version number, must be quoted`;
|
|
1539
|
+
}
|
|
1540
|
+
|
|
1541
|
+
if (/^-(?:\D|$)/v.test(text)) {
|
|
1542
|
+
if (/^-nan/iv.test(text)) {
|
|
1543
|
+
return 'NaN is not representable. Use null for a missing value';
|
|
1544
|
+
}
|
|
1545
|
+
|
|
1546
|
+
return /^-inf/iv.test(text) ? `"${abbreviate(text)}" is not a value. Negative infinity is written -infinity` : 'Expected a digit or "infinity" after "-"';
|
|
1547
|
+
}
|
|
1548
|
+
|
|
1549
|
+
return /[A-Za-z]/v.test(text) ? `Invalid number "${abbreviate(text)}". A string value must be quoted, as in '${abbreviate(text)}'` : `Invalid number "${abbreviate(text)}"`;
|
|
1550
|
+
}
|
|
1551
|
+
|
|
1552
|
+
function describeBadInstant(text) {
|
|
1553
|
+
if (text.length > MAX_DIAGNOSED_LENGTH) {
|
|
1554
|
+
return `Invalid instant "${abbreviate(text)}". An instant is written as 2026-09-19T14:00:00Z, with an optional fraction of up to nine digits and an offset of Z or ±HH:MM`;
|
|
1555
|
+
}
|
|
1556
|
+
|
|
1557
|
+
if (/^\d{4}-\d{2}-\d{2}$/v.test(text)) {
|
|
1558
|
+
return `${text} is a date, not an instant. An instant needs a time and an offset, as in ${text}T00:00:00Z; use a string for a date`;
|
|
1559
|
+
}
|
|
1560
|
+
|
|
1561
|
+
if (/^\d{4}-\d{2}-\d{2}t/v.test(text)) {
|
|
1562
|
+
return 'The date and time separator in an instant is an uppercase "T"';
|
|
1563
|
+
}
|
|
1564
|
+
|
|
1565
|
+
if (text.endsWith('z')) {
|
|
1566
|
+
return 'The UTC offset in an instant is an uppercase "Z"';
|
|
1567
|
+
}
|
|
1568
|
+
|
|
1569
|
+
if (/[+\-]\d{4}$/v.test(text)) {
|
|
1570
|
+
return 'An instant\'s offset is written with a colon, as in +07:00';
|
|
1571
|
+
}
|
|
1572
|
+
|
|
1573
|
+
return /^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d+)?$/v.test(text) ? 'An instant needs an offset: Z or ±HH:MM' : `Invalid instant "${abbreviate(text)}". An instant is written as 2026-09-19T14:00:00Z, with an optional fraction of up to nine digits and an offset of Z or ±HH:MM`;
|
|
1574
|
+
}
|