soml-lang 0.0.1 → 0.0.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/distribution/edit.d.ts +56 -0
- package/distribution/edit.js +635 -0
- package/distribution/error.d.ts +49 -0
- package/distribution/error.js +156 -0
- package/distribution/format.d.ts +62 -0
- package/distribution/format.js +403 -0
- package/distribution/index.d.ts +8 -0
- package/distribution/index.js +10 -0
- package/distribution/parse.d.ts +131 -0
- package/distribution/parse.js +1587 -0
- package/distribution/shared.d.ts +102 -0
- package/distribution/shared.js +285 -0
- package/distribution/stringify.d.ts +115 -0
- package/distribution/stringify.js +473 -0
- package/distribution/tree.d.ts +264 -0
- package/distribution/tree.js +468 -0
- package/package.json +34 -34
- package/readme.md +326 -39
- package/index.d.ts +0 -151
- package/index.js +0 -3
- package/source/error.js +0 -130
- package/source/parse.js +0 -1574
- package/source/shared.js +0 -166
- package/source/stringify.js +0 -439
|
@@ -0,0 +1,1587 @@
|
|
|
1
|
+
import { ParseError } from "./error.js";
|
|
2
|
+
import { formatKey } from "./stringify.js";
|
|
3
|
+
import { MAX_DEPTH, INT64_MIN, INT64_MAX, MIN_INSTANT, MAX_INSTANT, DURATION_UNITS, createDuration, requireTemporal, trimTrailingZeros, getIntegersOption, describeCharacter, formatCodePoint, describeKey, isBareKeyCharacter, isSpace, skipSpaces, skipSpacesBack, isBlankLine, findNumberEnd, findLineEnd, findBlockStringEnd, abbreviate, LF, SPACE, DOUBLE_QUOTE, HASH, SINGLE_QUOTE, ASTERISK, COMMA, DASH, DOT, SLASH, COLON, OPEN_BRACKET, CLOSE_BRACKET, OPEN_BRACE, CLOSE_BRACE, } from "./shared.js";
|
|
4
|
+
const valueEndCharacters = new Uint8Array(128);
|
|
5
|
+
for (const character of ' \t\n,]}#/') {
|
|
6
|
+
valueEndCharacters[character.codePointAt(0)] = 1;
|
|
7
|
+
}
|
|
8
|
+
/*
|
|
9
|
+
Whether a UTF-16 code unit may directly follow a scalar value: whitespace, a separator, a closing bracket, a comment, or the end, where `charCodeAt()` returns `NaN`.
|
|
10
|
+
*/
|
|
11
|
+
function isValueEnd(code) {
|
|
12
|
+
return Number.isNaN(code) || (code < 128 && valueEndCharacters[code] === 1);
|
|
13
|
+
}
|
|
14
|
+
/*
|
|
15
|
+
Whether a UTF-16 code unit is a space, a tab, a line feed, or the end.
|
|
16
|
+
*/
|
|
17
|
+
function isSpaceOrLineEnd(code) {
|
|
18
|
+
return isSpace(code) || code === LF || Number.isNaN(code);
|
|
19
|
+
}
|
|
20
|
+
/*
|
|
21
|
+
The lookup tables are maps rather than objects, so that a property added to `Object.prototype` is never read as an entry.
|
|
22
|
+
*/
|
|
23
|
+
const RADIX_DIGIT = new Map([
|
|
24
|
+
['x', code => (code >= 0x30 && code <= 0x39) || (code >= 0x41 && code <= 0x46)],
|
|
25
|
+
['o', code => code >= 0x30 && code <= 0x37],
|
|
26
|
+
['b', code => code === 0x30 || code === 0x31],
|
|
27
|
+
]);
|
|
28
|
+
/*
|
|
29
|
+
The longest token that the instant regular expression and the diagnostic regular expressions look at. Their repeated groups need stack in proportion to the input, so a longer token is rejected with a general message, or described from its first part.
|
|
30
|
+
*/
|
|
31
|
+
const MAX_DIAGNOSED_LENGTH = 1000;
|
|
32
|
+
const RADIX_NAME = { x: 'hexadecimal', o: 'octal', b: 'binary' };
|
|
33
|
+
const RADIX_ARTICLE = { x: 'A', o: 'An', b: 'A' };
|
|
34
|
+
const INSTANT_PREFIX = /^\d{4}-\d{2}-\d{2}/v;
|
|
35
|
+
const DURATION_UNIT_NAMES = DURATION_UNITS.keys().toArray();
|
|
36
|
+
const INSTANT = /^(?<year>\d{4})-(?<month>\d{2})-(?<day>\d{2})T(?<hour>\d{2}):(?<minute>\d{2}):(?<second>\d{2})(?:\.(?<fraction>\d+))?(?<offset>Z|[+\-](?<offsetHour>\d{2}):(?<offsetMinute>\d{2}))$/v;
|
|
37
|
+
const LOCAL_DATE_TIME = /^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d+)?$/v;
|
|
38
|
+
const INSTANT_FORMAT = 'An instant is written as 2026-09-19T14:00:00Z, with an optional fraction of up to nine digits and an offset of Z or ±HH:MM';
|
|
39
|
+
const UNICODE_ESCAPE = /u\{(?<hex>[\da-f]{1,6})\}/vy;
|
|
40
|
+
/*
|
|
41
|
+
Every C0 control character except tab and line feed, and DEL. They are errors anywhere in a document.
|
|
42
|
+
*/
|
|
43
|
+
// eslint-disable-next-line no-control-regex, regexp/no-control-character -- Control characters are exactly what must be found.
|
|
44
|
+
const CONTROL_CHARACTER = /[\u{0}-\u{8}\u{B}-\u{1F}\u{7F}]/v;
|
|
45
|
+
const LITERAL_STRING_END = /[\n']/gv;
|
|
46
|
+
const KEY_QUOTING_HINT = '. A key that contains characters other than letters, digits, “_”, and “-” must be quoted';
|
|
47
|
+
/*
|
|
48
|
+
The characters that may be meant as part of a key: visible ones that have no meaning in the grammar, which includes every bare key character. A `=` is left out, because after a key, as in `name=foo`, it was most likely meant as the `:` of INI and TOML, so a key that holds a `=` gets no quoting hint.
|
|
49
|
+
*/
|
|
50
|
+
const QUOTABLE_KEY_CHARACTER = String.raw `[^\p{Default_Ignorable_Code_Point}\p{Other}\p{White_Space}"#'*,.\/:=\[\]\{\}]`;
|
|
51
|
+
const QUOTABLE_KEY_START = new RegExp(`^${QUOTABLE_KEY_CHARACTER}`, 'v');
|
|
52
|
+
/*
|
|
53
|
+
A key made of them, up to its `:`, so the quoting hint can show it quoted.
|
|
54
|
+
*/
|
|
55
|
+
const QUOTABLE_KEY = new RegExp(`^${QUOTABLE_KEY_CHARACTER}+(?=:)`, 'v');
|
|
56
|
+
const ESCAPED_STRING_SPECIAL = /[\n"\\]/gv;
|
|
57
|
+
const SIMPLE_ESCAPES = new Map([
|
|
58
|
+
['\\', '\\'],
|
|
59
|
+
['"', '"'],
|
|
60
|
+
['n', '\n'],
|
|
61
|
+
['t', '\t'],
|
|
62
|
+
]);
|
|
63
|
+
/*
|
|
64
|
+
What a backslash followed by one of these characters was probably meant to be.
|
|
65
|
+
*/
|
|
66
|
+
const ESCAPE_MISTAKES = new Map([
|
|
67
|
+
['r', String.raw `There is no \r escape, because a carriage return cannot be represented`],
|
|
68
|
+
['\'', 'A \' needs no escape inside "..."'],
|
|
69
|
+
['\n', String.raw `A backslash must be followed by an escape character. Use \\ for a literal backslash, or a '...' string`],
|
|
70
|
+
]);
|
|
71
|
+
export function parse(text, options) {
|
|
72
|
+
const integers = getIntegersOption(options);
|
|
73
|
+
const source = decode(text);
|
|
74
|
+
checkCharacters(source);
|
|
75
|
+
// The default `createTime` makes a `Temporal` object of every instant and duration, so no `Time` is left.
|
|
76
|
+
return new Parser(source, integers).parseDocument();
|
|
77
|
+
}
|
|
78
|
+
/*
|
|
79
|
+
An instant, as nanoseconds since the Unix epoch, or a duration, as its length in nanoseconds. A class, so that a dotted key sees it as a value and not as an object.
|
|
80
|
+
*/
|
|
81
|
+
export class Time {
|
|
82
|
+
type;
|
|
83
|
+
nanoseconds;
|
|
84
|
+
constructor(type, nanoseconds) {
|
|
85
|
+
this.type = type;
|
|
86
|
+
this.nanoseconds = nanoseconds;
|
|
87
|
+
}
|
|
88
|
+
toTemporal() {
|
|
89
|
+
return this.type === 'Instant' ? new (requireTemporal().Instant)(this.nanoseconds) : createDuration(this.nanoseconds);
|
|
90
|
+
}
|
|
91
|
+
}
|
|
92
|
+
/*
|
|
93
|
+
The same as `parse()` for a string, except that each instant and duration is a `Time` instead of a `Temporal` object, so that it works without `Temporal`. For the tree, which makes the `Temporal` object only when the value is read.
|
|
94
|
+
*/
|
|
95
|
+
export function parseWithTimes(text) {
|
|
96
|
+
checkCharacters(text);
|
|
97
|
+
return new Parser(text, 'bigint', time => time).parseDocument();
|
|
98
|
+
}
|
|
99
|
+
const typedArrayTag = Object.getOwnPropertyDescriptor(Object.getPrototypeOf(Uint8Array.prototype), Symbol.toStringTag).get;
|
|
100
|
+
function decode(text) {
|
|
101
|
+
if (typeof text === 'string') {
|
|
102
|
+
return text;
|
|
103
|
+
}
|
|
104
|
+
// A brand check rather than `instanceof`, so that bytes from another realm, such as a `vm` context, are accepted too. The typed array tag getter reads an internal slot, so it cannot be faked with `Symbol.toStringTag`.
|
|
105
|
+
if (typedArrayTag.call(text) === 'Uint8Array') {
|
|
106
|
+
const bytes = text;
|
|
107
|
+
try {
|
|
108
|
+
// `ignoreBOM` keeps a BOM in the output, so that it can be rejected rather than silently dropped.
|
|
109
|
+
return new TextDecoder('utf-8', { fatal: true, ignoreBOM: true }).decode(bytes); // eslint-disable-line @typescript-eslint/naming-convention
|
|
110
|
+
}
|
|
111
|
+
catch (error) {
|
|
112
|
+
const offset = findInvalidUtf8(bytes);
|
|
113
|
+
// Every byte is valid, so the decoder failed for another reason, such as input longer than the longest possible string.
|
|
114
|
+
if (offset === bytes.length) {
|
|
115
|
+
throw error;
|
|
116
|
+
}
|
|
117
|
+
const valid = new TextDecoder('utf-8', { ignoreBOM: true }).decode(bytes.subarray(0, offset)); // eslint-disable-line @typescript-eslint/naming-convention
|
|
118
|
+
// A sequence that the end of the input cuts short decodes without error when more bytes may follow.
|
|
119
|
+
if (isTruncatedUtf8(bytes.subarray(offset))) {
|
|
120
|
+
throw ParseError.create('Incomplete UTF-8 sequence at the end of the input', valid, valid.length);
|
|
121
|
+
}
|
|
122
|
+
throw ParseError.create(`Invalid UTF-8 byte 0x${bytes[offset].toString(16).toUpperCase().padStart(2, '0')}`, valid, valid.length);
|
|
123
|
+
}
|
|
124
|
+
}
|
|
125
|
+
throw new TypeError(`Expected a string or a Uint8Array, got ${typeof text === 'object' ? (text === null ? 'null' : 'an object') : typeof text}`);
|
|
126
|
+
}
|
|
127
|
+
/*
|
|
128
|
+
Finds the first byte of the first invalid sequence. Only called on the error path, so each multi-byte sequence is checked with a fatal decoder rather than by reimplementing the UTF-8 rules for overlong forms, surrogates, and the U+10FFFF limit.
|
|
129
|
+
*/
|
|
130
|
+
function findInvalidUtf8(bytes) {
|
|
131
|
+
const decoder = new TextDecoder('utf-8', { fatal: true, ignoreBOM: true }); // eslint-disable-line @typescript-eslint/naming-convention
|
|
132
|
+
let index = 0;
|
|
133
|
+
while (index < bytes.length) {
|
|
134
|
+
const byte = bytes[index];
|
|
135
|
+
if (byte < 0x80) {
|
|
136
|
+
index++;
|
|
137
|
+
continue;
|
|
138
|
+
}
|
|
139
|
+
const length = byte >= 0xF0 ? 4 : (byte >= 0xE0 ? 3 : 2);
|
|
140
|
+
const sequence = bytes.subarray(index, index + length);
|
|
141
|
+
// A fatal decode also throws for a sequence that the end of the input cuts short.
|
|
142
|
+
try {
|
|
143
|
+
decoder.decode(sequence);
|
|
144
|
+
}
|
|
145
|
+
catch {
|
|
146
|
+
return index;
|
|
147
|
+
}
|
|
148
|
+
index += length;
|
|
149
|
+
}
|
|
150
|
+
return index;
|
|
151
|
+
}
|
|
152
|
+
function isTruncatedUtf8(bytes) {
|
|
153
|
+
try {
|
|
154
|
+
new TextDecoder('utf-8', { fatal: true }).decode(bytes, { stream: true });
|
|
155
|
+
return true;
|
|
156
|
+
}
|
|
157
|
+
catch {
|
|
158
|
+
return false;
|
|
159
|
+
}
|
|
160
|
+
}
|
|
161
|
+
/*
|
|
162
|
+
Characters that are errors wherever they appear, so they are checked once, up front.
|
|
163
|
+
*/
|
|
164
|
+
function checkCharacters(source) {
|
|
165
|
+
if (source.charCodeAt(0) === 0xFE_FF) {
|
|
166
|
+
throw ParseError.create('A byte order mark (BOM) is not allowed', source, 0);
|
|
167
|
+
}
|
|
168
|
+
const control = CONTROL_CHARACTER.exec(source);
|
|
169
|
+
if (control !== null) {
|
|
170
|
+
const { index } = control;
|
|
171
|
+
const code = source.charCodeAt(index);
|
|
172
|
+
if (code === 0x0D) {
|
|
173
|
+
throw ParseError.create('A carriage return (U+000D) is not allowed anywhere. Use LF line endings', source, index);
|
|
174
|
+
}
|
|
175
|
+
const escape = String.raw `\u{${code.toString(16)}}`;
|
|
176
|
+
throw ParseError.create(`A raw control character (${formatCodePoint(code)}) is not allowed anywhere, including in strings and comments. In a string, write it as the escape ${escape} inside "..."`, source, index);
|
|
177
|
+
}
|
|
178
|
+
if (source.isWellFormed()) {
|
|
179
|
+
return;
|
|
180
|
+
}
|
|
181
|
+
// With the `v` flag, a surrogate in the class matches only when it is unpaired.
|
|
182
|
+
const offset = /[\u{D800}-\u{DFFF}]/v.exec(source).index;
|
|
183
|
+
throw ParseError.create(`A lone surrogate (${describeCharacter(source.charCodeAt(offset))}) is not a Unicode scalar value`, source, offset);
|
|
184
|
+
}
|
|
185
|
+
/*
|
|
186
|
+
Creates an own property without ever invoking a setter, like `JSON.parse`. Plain assignment is used when `Object.prototype` has no property of that name, which is the fast common case. Otherwise, as for `__proto__`, `toString`, or anything a library added to the prototype, assignment would call a setter or fail on a frozen prototype.
|
|
187
|
+
*/
|
|
188
|
+
function defineMember(object, key, value) {
|
|
189
|
+
if (Reflect.has(Object.prototype, key)) {
|
|
190
|
+
Object.defineProperty(object, key, {
|
|
191
|
+
value,
|
|
192
|
+
writable: true,
|
|
193
|
+
enumerable: true,
|
|
194
|
+
configurable: true,
|
|
195
|
+
});
|
|
196
|
+
}
|
|
197
|
+
else {
|
|
198
|
+
object[key] = value;
|
|
199
|
+
}
|
|
200
|
+
}
|
|
201
|
+
function describeValue(value) {
|
|
202
|
+
if (Array.isArray(value)) {
|
|
203
|
+
return 'an array';
|
|
204
|
+
}
|
|
205
|
+
// Every parsed object is created with `{}`, unlike an instant or a duration.
|
|
206
|
+
return value !== null && typeof value === 'object' && Object.getPrototypeOf(value) === Object.prototype ? 'an object' : 'a value';
|
|
207
|
+
}
|
|
208
|
+
function formatPath(path) {
|
|
209
|
+
return path.map(segment => describeKey(segment)).join('.');
|
|
210
|
+
}
|
|
211
|
+
class Parser {
|
|
212
|
+
#source;
|
|
213
|
+
#index = 0;
|
|
214
|
+
#integers;
|
|
215
|
+
#createTime;
|
|
216
|
+
/*
|
|
217
|
+
Objects created by dotted keys. They may be extended by further dotted keys, while an object written with braces is closed.
|
|
218
|
+
*/
|
|
219
|
+
#dottedObjects = new WeakSet();
|
|
220
|
+
/*
|
|
221
|
+
Where the document's collection starts, after the comments and whitespace before it.
|
|
222
|
+
*/
|
|
223
|
+
#documentStart = 0;
|
|
224
|
+
constructor(source, integers, createTime = time => time.toTemporal()) {
|
|
225
|
+
this.#source = source;
|
|
226
|
+
this.#integers = integers;
|
|
227
|
+
this.#createTime = createTime;
|
|
228
|
+
}
|
|
229
|
+
#fail(reason, offset = this.#index) {
|
|
230
|
+
throw ParseError.create(reason, this.#source, offset);
|
|
231
|
+
}
|
|
232
|
+
#code(offset = this.#index) {
|
|
233
|
+
return this.#source.charCodeAt(offset);
|
|
234
|
+
}
|
|
235
|
+
#isAtEnd() {
|
|
236
|
+
return this.#index >= this.#source.length;
|
|
237
|
+
}
|
|
238
|
+
/*
|
|
239
|
+
`bare-object = entry ( entry-sep entry )*`, where a separator is one or more line breaks.
|
|
240
|
+
*/
|
|
241
|
+
#parseBareObject() {
|
|
242
|
+
const object = {};
|
|
243
|
+
const start = this.#index;
|
|
244
|
+
try {
|
|
245
|
+
this.#parseEntry(object, 1);
|
|
246
|
+
}
|
|
247
|
+
catch (error) {
|
|
248
|
+
if (error instanceof ParseError) {
|
|
249
|
+
this.#diagnoseBareValue(start);
|
|
250
|
+
}
|
|
251
|
+
throw error;
|
|
252
|
+
}
|
|
253
|
+
let hasLineBreak = this.#skipTrivia();
|
|
254
|
+
while (!this.#isAtEnd()) {
|
|
255
|
+
if (this.#code() === COMMA) {
|
|
256
|
+
this.#fail('Top-level entries are separated by line breaks, not commas. Use braces for a one-line object');
|
|
257
|
+
}
|
|
258
|
+
if (!hasLineBreak) {
|
|
259
|
+
this.#fail(`Expected a line break before the next entry, but found ${this.#describeHere()}${this.#slashCommentHint()}`);
|
|
260
|
+
}
|
|
261
|
+
this.#parseEntry(object, 1);
|
|
262
|
+
hasLineBreak = this.#skipTrivia();
|
|
263
|
+
}
|
|
264
|
+
return object;
|
|
265
|
+
}
|
|
266
|
+
/*
|
|
267
|
+
A whole document that is one scalar, such as `5` or an instant, fails as an entry. That failure is reported as what it is.
|
|
268
|
+
*/
|
|
269
|
+
#diagnoseBareValue(start) {
|
|
270
|
+
this.#index = start;
|
|
271
|
+
let isBareValue = false;
|
|
272
|
+
try {
|
|
273
|
+
this.#parseValue(1);
|
|
274
|
+
this.#skipTrivia();
|
|
275
|
+
isBareValue = this.#isAtEnd();
|
|
276
|
+
}
|
|
277
|
+
catch {
|
|
278
|
+
// Not a bare value either, so the original error stands.
|
|
279
|
+
}
|
|
280
|
+
if (isBareValue) {
|
|
281
|
+
this.#fail('A bare value is not a document. A document is an object or an array, so write it as `key: value` or `[value]`', start);
|
|
282
|
+
}
|
|
283
|
+
}
|
|
284
|
+
/*
|
|
285
|
+
Skips spaces, tabs, line breaks, and comments, and reports whether a line break was crossed. A line break inside a block comment does not count.
|
|
286
|
+
*/
|
|
287
|
+
#skipTrivia() {
|
|
288
|
+
const source = this.#source;
|
|
289
|
+
let hasCrossedLineBreak = false;
|
|
290
|
+
for (;;) {
|
|
291
|
+
const code = source.charCodeAt(this.#index);
|
|
292
|
+
if (isSpace(code)) {
|
|
293
|
+
this.#index++;
|
|
294
|
+
}
|
|
295
|
+
else if (code === LF) {
|
|
296
|
+
hasCrossedLineBreak = true;
|
|
297
|
+
this.#index++;
|
|
298
|
+
}
|
|
299
|
+
else if (!this.#skipComment()) {
|
|
300
|
+
return hasCrossedLineBreak;
|
|
301
|
+
}
|
|
302
|
+
}
|
|
303
|
+
}
|
|
304
|
+
/*
|
|
305
|
+
Skips the comment that starts here, or returns `false` when there is none.
|
|
306
|
+
*/
|
|
307
|
+
#skipComment() {
|
|
308
|
+
const source = this.#source;
|
|
309
|
+
const code = source.charCodeAt(this.#index);
|
|
310
|
+
if (code === HASH) {
|
|
311
|
+
this.#index = findLineEnd(source, this.#index);
|
|
312
|
+
return true;
|
|
313
|
+
}
|
|
314
|
+
if (code === SLASH && source.charCodeAt(this.#index + 1) === ASTERISK) {
|
|
315
|
+
this.#skipBlockComment();
|
|
316
|
+
return true;
|
|
317
|
+
}
|
|
318
|
+
return false;
|
|
319
|
+
}
|
|
320
|
+
#skipBlockComment() {
|
|
321
|
+
const source = this.#source;
|
|
322
|
+
const start = this.#index;
|
|
323
|
+
const end = source.indexOf('*/', start + 2);
|
|
324
|
+
if (end === -1) {
|
|
325
|
+
this.#fail('Unterminated block comment', start);
|
|
326
|
+
}
|
|
327
|
+
const nested = source.indexOf('/*', start + 2);
|
|
328
|
+
// The body may not contain `/*`. An opening that overlaps the closing `*/`, as in `/*/`, is not inside the body.
|
|
329
|
+
if (nested !== -1 && nested + 2 <= end) {
|
|
330
|
+
this.#fail('Block comments cannot be nested, and their body may not contain “/*”', nested);
|
|
331
|
+
}
|
|
332
|
+
this.#index = end + 2;
|
|
333
|
+
}
|
|
334
|
+
/*
|
|
335
|
+
`entry = key ":" ws value`.
|
|
336
|
+
*/
|
|
337
|
+
#parseEntry(object, depth) {
|
|
338
|
+
const keyStart = this.#index;
|
|
339
|
+
const path = this.#parseKey();
|
|
340
|
+
if (depth + path.length - 1 > MAX_DEPTH) {
|
|
341
|
+
this.#failTooDeep(keyStart);
|
|
342
|
+
}
|
|
343
|
+
const code = this.#code();
|
|
344
|
+
if (code !== COLON) {
|
|
345
|
+
this.#failMissingColon(keyStart);
|
|
346
|
+
}
|
|
347
|
+
const colon = this.#index;
|
|
348
|
+
this.#index++;
|
|
349
|
+
this.#skipTrivia();
|
|
350
|
+
let value;
|
|
351
|
+
try {
|
|
352
|
+
value = this.#parseValue(depth + path.length);
|
|
353
|
+
}
|
|
354
|
+
catch (error) {
|
|
355
|
+
if (error instanceof ParseError) {
|
|
356
|
+
this.#diagnoseBadValue(error, path, keyStart, colon);
|
|
357
|
+
}
|
|
358
|
+
throw error;
|
|
359
|
+
}
|
|
360
|
+
this.#assign(object, path, value, keyStart);
|
|
361
|
+
}
|
|
362
|
+
/*
|
|
363
|
+
Reports a value that failed to parse as what the entry was meant to be, when that is clear.
|
|
364
|
+
*/
|
|
365
|
+
#diagnoseBadValue(error, path, keyStart, colon) {
|
|
366
|
+
// A key that contains a `:`, as in `12:30: 'lunch'`, ends at the first `:`, and the rest is read as the value.
|
|
367
|
+
if (isBareKeyCharacter(this.#code(colon + 1))) {
|
|
368
|
+
this.#diagnoseKeyWithColon(keyStart);
|
|
369
|
+
}
|
|
370
|
+
this.#index = colon + 1;
|
|
371
|
+
const isValueOnNextLine = this.#skipTrivia();
|
|
372
|
+
const valueStart = this.#index;
|
|
373
|
+
const commentHint = this.#describeCommentAsValue(colon);
|
|
374
|
+
if (isValueOnNextLine) {
|
|
375
|
+
this.#diagnoseMissingValue(valueStart, path, keyStart, commentHint);
|
|
376
|
+
}
|
|
377
|
+
// A value that was left out at the end of the document or of an object.
|
|
378
|
+
const code = this.#code(valueStart);
|
|
379
|
+
if (commentHint !== '' && error.offset === valueStart && (code === CLOSE_BRACE || Number.isNaN(code))) {
|
|
380
|
+
this.#fail(`${error.reason}${commentHint}`, valueStart);
|
|
381
|
+
}
|
|
382
|
+
}
|
|
383
|
+
/*
|
|
384
|
+
A bare key followed directly by `:` and more of a key, up to a `:` that ends the key, as in `a:b: 1`.
|
|
385
|
+
*/
|
|
386
|
+
#diagnoseKeyWithColon(keyStart) {
|
|
387
|
+
const source = this.#source;
|
|
388
|
+
for (let index = keyStart; index < keyStart + MAX_DIAGNOSED_LENGTH; index++) {
|
|
389
|
+
const code = source.charCodeAt(index);
|
|
390
|
+
if (code === COLON) {
|
|
391
|
+
const next = source.charCodeAt(index + 1);
|
|
392
|
+
if (next === LF || isSpace(next) || Number.isNaN(next)) {
|
|
393
|
+
this.#fail(`A key that contains “:” must be quoted, as in '${abbreviate(source.slice(keyStart, index))}'`, keyStart);
|
|
394
|
+
}
|
|
395
|
+
}
|
|
396
|
+
else if (!isBareKeyCharacter(code)) {
|
|
397
|
+
return;
|
|
398
|
+
}
|
|
399
|
+
}
|
|
400
|
+
}
|
|
401
|
+
/*
|
|
402
|
+
The hint for a `#` directly after a `:`, as in `color: #FFF`, which starts a comment rather than a value.
|
|
403
|
+
*/
|
|
404
|
+
#describeCommentAsValue(colon) {
|
|
405
|
+
const source = this.#source;
|
|
406
|
+
const hash = skipSpaces(source, colon + 1);
|
|
407
|
+
if (source.charCodeAt(hash) !== HASH) {
|
|
408
|
+
return '';
|
|
409
|
+
}
|
|
410
|
+
// A `#` followed by a space, or by another `#`, starts an ordinary comment. A separator or a closing bracket after the value is not part of it. The regular expression, which needs stack in proportion to its match, only sees the start of a long comment, which is cut short in the message anyway.
|
|
411
|
+
const match = /^#[^\t\n #,\]\}][^\t\n ,\]\}]*/v.exec(source.slice(hash, hash + MAX_DIAGNOSED_LENGTH));
|
|
412
|
+
return match === null ? '' : `. “#” starts a comment, so a value that starts with “#” must be quoted${quotingExample(match[0])}`;
|
|
413
|
+
}
|
|
414
|
+
/*
|
|
415
|
+
An entry whose value was left out, as in `a:` followed by `'b': 1` on the next line, reads the next key as the value. That failure is reported as what it is.
|
|
416
|
+
*/
|
|
417
|
+
#diagnoseMissingValue(start, path, keyStart, commentHint) {
|
|
418
|
+
// A YAML block sequence.
|
|
419
|
+
if (this.#code(start) === DASH && isSpace(this.#code(start + 1))) {
|
|
420
|
+
this.#fail('Expected a value, but found a “-” list. An array is written in brackets, as in [80, 443]', start);
|
|
421
|
+
}
|
|
422
|
+
this.#index = start;
|
|
423
|
+
let key;
|
|
424
|
+
try {
|
|
425
|
+
key = this.#parseKey();
|
|
426
|
+
}
|
|
427
|
+
catch {
|
|
428
|
+
// Not a key either, so the original error stands.
|
|
429
|
+
}
|
|
430
|
+
if (key === undefined || this.#code() !== COLON) {
|
|
431
|
+
return;
|
|
432
|
+
}
|
|
433
|
+
// Whitespace or the end follows the `:` of a bare key. A digit follows the `:` in an instant such as `2026-09-19T25:00:00Z`, whose own error is more precise. A quoted key cannot be part of a value, so anything may follow its `:`.
|
|
434
|
+
const next = this.#code(this.#index + 1);
|
|
435
|
+
const firstCode = this.#code(start);
|
|
436
|
+
if (firstCode !== SINGLE_QUOTE && firstCode !== DOUBLE_QUOTE && next !== LF && !isSpace(next) && !Number.isNaN(next)) {
|
|
437
|
+
return;
|
|
438
|
+
}
|
|
439
|
+
const hint = commentHint === '' ? this.#describeIndentedKey(path, key, keyStart, start) : commentHint;
|
|
440
|
+
this.#fail(`Expected a value, but found the key ${abbreviate(formatPath(key), 200)}${hint}`, start);
|
|
441
|
+
}
|
|
442
|
+
/*
|
|
443
|
+
The hint for a key at `start` that is indented under the entry whose value is missing, as YAML nests an object.
|
|
444
|
+
*/
|
|
445
|
+
#describeIndentedKey(path, key, keyStart, start) {
|
|
446
|
+
const keyIndentation = getIndentation(this.#source, keyStart);
|
|
447
|
+
const indentation = getIndentation(this.#source, start);
|
|
448
|
+
if (keyIndentation === undefined || indentation === undefined || indentation <= keyIndentation) {
|
|
449
|
+
return '';
|
|
450
|
+
}
|
|
451
|
+
// The keys are written as in a document, so that the suggestion is valid, rather than as JSON strings like the rest of the message, whose escapes, such as `\b`, are not all valid.
|
|
452
|
+
const parent = abbreviate(path.map(segment => formatKey(segment)).join('.'), 200);
|
|
453
|
+
const child = abbreviate(key.map(segment => formatKey(segment)).join('.'), 200);
|
|
454
|
+
return `. Indentation does not nest objects, so write ${parent}: {${child}: …} or ${parent}.${child}: …`;
|
|
455
|
+
}
|
|
456
|
+
#failMissingColon(keyStart) {
|
|
457
|
+
const source = this.#source;
|
|
458
|
+
const next = skipSpaces(source, this.#index);
|
|
459
|
+
const nextCode = source.charCodeAt(next);
|
|
460
|
+
if (nextCode === COLON && next > this.#index) {
|
|
461
|
+
this.#fail('Whitespace is not allowed between a key and its “:”');
|
|
462
|
+
}
|
|
463
|
+
// A block comment followed by the ":" was meant to come before it.
|
|
464
|
+
if (nextCode === SLASH && source.charCodeAt(next + 1) === ASTERISK) {
|
|
465
|
+
const commentEnd = source.indexOf('*/', next + 2);
|
|
466
|
+
if (commentEnd !== -1 && source.charCodeAt(skipSpaces(source, commentEnd + 2)) === COLON) {
|
|
467
|
+
this.#fail('A comment is not allowed between a key and its “:”', next);
|
|
468
|
+
}
|
|
469
|
+
}
|
|
470
|
+
const colon = source.indexOf(':', next);
|
|
471
|
+
// A key with a space in it, such as `the name: 1`. The hint is only given for plain words, because quoting a dotted or quoted key would change what it means, and for a `:` that whitespace or the end follows, because quoting the words before the `:` in `server localhost:8080` would give a valid document with another meaning. So `the name:1` gets no hint.
|
|
472
|
+
if (colon !== -1 && next > this.#index && isBareKeyCharacter(nextCode) && colon < findLineEnd(source, next) && isSpaceOrLineEnd(source.charCodeAt(colon + 1))) {
|
|
473
|
+
// Only spaces and tabs are trimmed, because `trimEnd()` would also remove characters that are not whitespace in SOML, such as U+00A0.
|
|
474
|
+
const key = source.slice(keyStart, skipSpacesBack(source, colon));
|
|
475
|
+
if (isWordsWithSpaces(key)) {
|
|
476
|
+
this.#fail(`A bare key cannot contain spaces. Quote it, as in '${abbreviate(key)}'`, keyStart);
|
|
477
|
+
}
|
|
478
|
+
}
|
|
479
|
+
if (this.#isAtEnd() || this.#code() === LF) {
|
|
480
|
+
this.#fail('Expected “:” after the key');
|
|
481
|
+
}
|
|
482
|
+
// A character directly after a bare key is most likely meant to be part of it.
|
|
483
|
+
const previousCode = source.charCodeAt(this.#index - 1);
|
|
484
|
+
const hint = previousCode !== SINGLE_QUOTE && previousCode !== DOUBLE_QUOTE && next === this.#index && QUOTABLE_KEY_START.test(source.slice(next, next + 2)) ? KEY_QUOTING_HINT : '';
|
|
485
|
+
this.#index = next;
|
|
486
|
+
this.#fail(`Expected “:” after the key, but found ${this.#describeHere()}${hint}`);
|
|
487
|
+
}
|
|
488
|
+
#parseKey() {
|
|
489
|
+
const path = [this.#parseKeySegment()];
|
|
490
|
+
while (this.#code() === DOT) {
|
|
491
|
+
this.#index++;
|
|
492
|
+
path.push(this.#parseKeySegment(true));
|
|
493
|
+
}
|
|
494
|
+
return path;
|
|
495
|
+
}
|
|
496
|
+
#parseKeySegment(isAfterDot = false) {
|
|
497
|
+
const source = this.#source;
|
|
498
|
+
const start = this.#index;
|
|
499
|
+
const code = source.charCodeAt(start);
|
|
500
|
+
if (code === SINGLE_QUOTE || code === DOUBLE_QUOTE) {
|
|
501
|
+
if (source.charCodeAt(start + 1) === code && source.charCodeAt(start + 2) === code) {
|
|
502
|
+
this.#fail('A block string cannot be a key');
|
|
503
|
+
}
|
|
504
|
+
// A key is a string, so it may hold anything a string can, including a line feed written as an escape.
|
|
505
|
+
return code === SINGLE_QUOTE ? this.#parseLiteralString() : this.#parseEscapedString();
|
|
506
|
+
}
|
|
507
|
+
let end = start;
|
|
508
|
+
while (isBareKeyCharacter(source.charCodeAt(end))) {
|
|
509
|
+
end++;
|
|
510
|
+
}
|
|
511
|
+
if (end === start) {
|
|
512
|
+
if (code === OPEN_BRACKET) {
|
|
513
|
+
this.#diagnoseTableHeader();
|
|
514
|
+
}
|
|
515
|
+
const expected = isAfterDot ? 'Expected a key segment after “.”' : 'Expected a key';
|
|
516
|
+
this.#fail(this.#isAtEnd() ? expected : `${expected}, but found ${this.#describeHere()}${this.#keyQuotingHint()}${this.#slashCommentHint()}`);
|
|
517
|
+
}
|
|
518
|
+
this.#index = end;
|
|
519
|
+
return source.slice(start, end);
|
|
520
|
+
}
|
|
521
|
+
/*
|
|
522
|
+
The hint for a key that starts with a character a bare key cannot hold, such as the `$` in `$schema`, or nothing for a character with another meaning, such as `}`.
|
|
523
|
+
*/
|
|
524
|
+
#keyQuotingHint() {
|
|
525
|
+
if (!QUOTABLE_KEY_START.test(this.#source.slice(this.#index, this.#index + 2))) {
|
|
526
|
+
return '';
|
|
527
|
+
}
|
|
528
|
+
// A longer key gets the hint without the example, and the regular expression, which backtracks, never sees a huge run.
|
|
529
|
+
const key = QUOTABLE_KEY.exec(this.#source.slice(this.#index, this.#index + MAX_DIAGNOSED_LENGTH))?.[0];
|
|
530
|
+
return key === undefined ? KEY_QUOTING_HINT : `${KEY_QUOTING_HINT}, as in '${abbreviate(key)}'`;
|
|
531
|
+
}
|
|
532
|
+
#assign(object, path, value, keyStart) {
|
|
533
|
+
let target = object;
|
|
534
|
+
const lastIndex = path.length - 1;
|
|
535
|
+
for (let index = 0; index < lastIndex; index++) {
|
|
536
|
+
const key = path[index];
|
|
537
|
+
if (!Object.hasOwn(target, key)) {
|
|
538
|
+
const child = {};
|
|
539
|
+
this.#dottedObjects.add(child);
|
|
540
|
+
defineMember(target, key, child);
|
|
541
|
+
target = child;
|
|
542
|
+
continue;
|
|
543
|
+
}
|
|
544
|
+
// Only an object built by dotted keys is in the set, which is checked next.
|
|
545
|
+
const existing = target[key];
|
|
546
|
+
if (!this.#dottedObjects.has(existing)) {
|
|
547
|
+
const prefix = formatPath(path.slice(0, index + 1));
|
|
548
|
+
const what = describeValue(existing);
|
|
549
|
+
const reason = what === 'an object' ? 'an object written with braces, which is closed' : what;
|
|
550
|
+
this.#fail(`Cannot set ${formatPath(path)}, because ${prefix} is already ${reason} and a dotted key cannot extend it`, keyStart);
|
|
551
|
+
}
|
|
552
|
+
target = existing;
|
|
553
|
+
}
|
|
554
|
+
const key = path[lastIndex];
|
|
555
|
+
if (Object.hasOwn(target, key)) {
|
|
556
|
+
if (this.#dottedObjects.has(target[key])) {
|
|
557
|
+
this.#fail(`Cannot set ${formatPath(path)}: it is already an object built by dotted keys`, keyStart);
|
|
558
|
+
}
|
|
559
|
+
this.#fail(`Duplicate key ${formatPath(path)}`, keyStart);
|
|
560
|
+
}
|
|
561
|
+
defineMember(target, key, value);
|
|
562
|
+
}
|
|
563
|
+
#failTooDeep(offset = this.#index) {
|
|
564
|
+
this.#fail(`The document is nested more than ${MAX_DEPTH} levels deep`, offset);
|
|
565
|
+
}
|
|
566
|
+
#parseObject(depth) {
|
|
567
|
+
const object = {};
|
|
568
|
+
this.#parseItems(depth, CLOSE_BRACE, () => {
|
|
569
|
+
this.#parseEntry(object, depth);
|
|
570
|
+
});
|
|
571
|
+
return object;
|
|
572
|
+
}
|
|
573
|
+
#parseArray(depth) {
|
|
574
|
+
const array = [];
|
|
575
|
+
this.#parseItems(depth, CLOSE_BRACKET, () => {
|
|
576
|
+
array.push(this.#parseValue(depth + 1));
|
|
577
|
+
});
|
|
578
|
+
return array;
|
|
579
|
+
}
|
|
580
|
+
/*
|
|
581
|
+
`{` or `[` at depth `depth`, then items separated by a comma, a line break, or both, with an optional trailing comma, then the `closing` bracket. A comma must be on the line of the item before it.
|
|
582
|
+
*/
|
|
583
|
+
#parseItems(depth, closing, parseItem) {
|
|
584
|
+
if (depth > MAX_DEPTH) {
|
|
585
|
+
this.#failTooDeep();
|
|
586
|
+
}
|
|
587
|
+
const start = this.#index;
|
|
588
|
+
this.#index++;
|
|
589
|
+
this.#skipTrivia();
|
|
590
|
+
while (this.#code() !== closing) {
|
|
591
|
+
let hasLineBreak = false;
|
|
592
|
+
let itemEnd = this.#index;
|
|
593
|
+
if (!this.#isAtEnd()) {
|
|
594
|
+
parseItem();
|
|
595
|
+
itemEnd = this.#index;
|
|
596
|
+
hasLineBreak = this.#skipTrivia();
|
|
597
|
+
}
|
|
598
|
+
if (this.#isAtEnd()) {
|
|
599
|
+
this.#fail(`Unterminated ${closing === CLOSE_BRACE ? 'object' : 'array'}: expected “${String.fromCharCode(closing)}”`, start);
|
|
600
|
+
}
|
|
601
|
+
if (this.#code() === COMMA) {
|
|
602
|
+
if (hasLineBreak) {
|
|
603
|
+
this.#fail('A comma must be on the same line as the item before it. The line break already separates the items, so remove the comma');
|
|
604
|
+
}
|
|
605
|
+
this.#index++;
|
|
606
|
+
this.#skipTrivia();
|
|
607
|
+
}
|
|
608
|
+
else if (!hasLineBreak && this.#code() !== closing) {
|
|
609
|
+
const item = closing === CLOSE_BRACE ? 'an object member' : 'an array item';
|
|
610
|
+
// Without a line break that separates, a line break in the gap is inside a block comment.
|
|
611
|
+
const hint = this.#source.slice(itemEnd, this.#index).includes('\n') ? '. A line break inside a block comment does not separate items' : this.#slashCommentHint();
|
|
612
|
+
this.#fail(`Expected “,”, a line break, or “${String.fromCharCode(closing)}” after ${item}, but found ${this.#describeHere()}${hint}`);
|
|
613
|
+
}
|
|
614
|
+
}
|
|
615
|
+
this.#index++;
|
|
616
|
+
}
|
|
617
|
+
#parseValue(depth) {
|
|
618
|
+
const code = this.#code();
|
|
619
|
+
if (code === OPEN_BRACE) {
|
|
620
|
+
return this.#parseObject(depth);
|
|
621
|
+
}
|
|
622
|
+
if (code === OPEN_BRACKET) {
|
|
623
|
+
return this.#parseArray(depth);
|
|
624
|
+
}
|
|
625
|
+
const start = this.#index;
|
|
626
|
+
let value;
|
|
627
|
+
// An int or a float written with digits.
|
|
628
|
+
let isNumber = false;
|
|
629
|
+
if (code === SINGLE_QUOTE || code === DOUBLE_QUOTE) {
|
|
630
|
+
value = this.#parseString();
|
|
631
|
+
}
|
|
632
|
+
else if (code === 0x74 /* t */ && this.#isKeyword('true')) {
|
|
633
|
+
value = true;
|
|
634
|
+
}
|
|
635
|
+
else if (code === 0x66 /* f */ && this.#isKeyword('false')) {
|
|
636
|
+
value = false;
|
|
637
|
+
}
|
|
638
|
+
else if (code === 0x6E /* n */ && this.#isKeyword('null')) {
|
|
639
|
+
value = null;
|
|
640
|
+
}
|
|
641
|
+
else if (code === 0x69 /* i */ && this.#isKeyword('infinity')) {
|
|
642
|
+
value = Infinity;
|
|
643
|
+
}
|
|
644
|
+
else if (code === DASH && this.#isKeyword('-infinity')) {
|
|
645
|
+
value = -Infinity;
|
|
646
|
+
}
|
|
647
|
+
else if (code === DASH || isDigit(code)) {
|
|
648
|
+
value = this.#parseNumberOrInstant();
|
|
649
|
+
isNumber = typeof value === 'bigint' || typeof value === 'number';
|
|
650
|
+
}
|
|
651
|
+
else {
|
|
652
|
+
this.#failUnexpectedValue();
|
|
653
|
+
}
|
|
654
|
+
if (!isValueEnd(this.#code())) {
|
|
655
|
+
this.#failAfterValue(start, isNumber);
|
|
656
|
+
}
|
|
657
|
+
if (isNumber && isSpace(this.#code())) {
|
|
658
|
+
this.#diagnoseUnitAfterSpace(start);
|
|
659
|
+
}
|
|
660
|
+
return value;
|
|
661
|
+
}
|
|
662
|
+
/*
|
|
663
|
+
A character that cannot follow the value that starts at `start`.
|
|
664
|
+
*/
|
|
665
|
+
#failAfterValue(start, isNumber) {
|
|
666
|
+
const code = this.#code();
|
|
667
|
+
// Go writes microseconds as `µs`, with the micro sign or the Greek letter mu. The duration is only suggested when it is valid, which a number with an exponent or a radix prefix, for example, is not.
|
|
668
|
+
if (isNumber && (code === 0xB5 || code === 0x3_BC) && this.#code(this.#index + 1) === 0x73 /* s */) {
|
|
669
|
+
const number = this.#source.slice(start, this.#index);
|
|
670
|
+
this.#fail(`The unit for microseconds is written us${isValidValue(`${number}us`) ? `, as in ${abbreviate(number)}us` : ''}`);
|
|
671
|
+
}
|
|
672
|
+
// A `''` inside a '...' string, as SQL and YAML escape a quote, ends the string.
|
|
673
|
+
if (code === SINGLE_QUOTE && this.#code(start) === SINGLE_QUOTE) {
|
|
674
|
+
this.#fail('There is no \'\' escape in a \'...\' string. Write a string that contains \' as "...", as in "it\'s"');
|
|
675
|
+
}
|
|
676
|
+
this.#fail(`Unexpected ${this.#describeHere()} after a value`);
|
|
677
|
+
}
|
|
678
|
+
/*
|
|
679
|
+
A number followed by a space and a word, such as `512 MiB` or `10 seconds`, which is an error anyway. A word followed by more than spaces, a separator, or a comment is left to the general errors, because it may be a key, as in `a: 1 b: 2`, or a sentence.
|
|
680
|
+
*/
|
|
681
|
+
#diagnoseUnitAfterSpace(start) {
|
|
682
|
+
const source = this.#source;
|
|
683
|
+
const unitStart = skipSpaces(source, this.#index);
|
|
684
|
+
let unitEnd = unitStart;
|
|
685
|
+
while (isLetter(source.charCodeAt(unitEnd))) {
|
|
686
|
+
unitEnd++;
|
|
687
|
+
}
|
|
688
|
+
const unit = source.slice(unitStart, unitEnd);
|
|
689
|
+
// A keyword after a number is a missing comma.
|
|
690
|
+
if (unit === '' || !isValueEnd(source.charCodeAt(skipSpaces(source, unitEnd))) || ['true', 'false', 'null', 'infinity'].includes(unit)) {
|
|
691
|
+
return;
|
|
692
|
+
}
|
|
693
|
+
const number = source.slice(start, this.#index);
|
|
694
|
+
// The duration is only suggested when it is valid, which a number with an exponent or a radix prefix, a fraction of a nanosecond, a negative zero, or a value outside the 64-bit range is not. The string is always offered, because a unit such as `m` may mean meters rather than minutes.
|
|
695
|
+
const duration = DURATION_UNITS.has(unit) && isValidValue(`${number}${unit}`) ? `${abbreviate(number)}${unit}` : '10s';
|
|
696
|
+
this.#fail(`A unit cannot follow a number after a space. Write a duration without the space, as in ${duration}, and anything else, such as a size, as a string, as in '${abbreviate(`${number} ${unit}`)}'`, unitStart);
|
|
697
|
+
}
|
|
698
|
+
#isKeyword(word) {
|
|
699
|
+
if (!this.#source.startsWith(word, this.#index)) {
|
|
700
|
+
return false;
|
|
701
|
+
}
|
|
702
|
+
const next = this.#code(this.#index + word.length);
|
|
703
|
+
// A longer word, such as `nullable`, is not the keyword.
|
|
704
|
+
if (isBareKeyCharacter(next)) {
|
|
705
|
+
return false;
|
|
706
|
+
}
|
|
707
|
+
this.#index += word.length;
|
|
708
|
+
return true;
|
|
709
|
+
}
|
|
710
|
+
#failUnexpectedValue() {
|
|
711
|
+
if (this.#isAtEnd()) {
|
|
712
|
+
this.#fail('Expected a value, but reached the end of the document');
|
|
713
|
+
}
|
|
714
|
+
const code = this.#code();
|
|
715
|
+
if (code === 0x2B /* + */) {
|
|
716
|
+
this.#fail('A “+” sign is not allowed. A number without a sign is positive');
|
|
717
|
+
}
|
|
718
|
+
// A `.` that no digit follows begins a string, such as `.env` or `./foo`, rather than a number.
|
|
719
|
+
if (code === DOT) {
|
|
720
|
+
this.#fail(isDigit(this.#code(this.#index + 1)) ? 'A number cannot begin with “.”; write a digit before it, as in 0.5' : describeUnknownWord('.', this.#unquotedText()));
|
|
721
|
+
}
|
|
722
|
+
let wordEnd = this.#index;
|
|
723
|
+
while (isBareKeyCharacter(this.#code(wordEnd))) {
|
|
724
|
+
wordEnd++;
|
|
725
|
+
}
|
|
726
|
+
if (wordEnd > this.#index) {
|
|
727
|
+
const word = this.#source.slice(this.#index, wordEnd);
|
|
728
|
+
// A key where a value should be, as when the value of an entry is left out and the next line has a key, or as in `[a: 1]`, is reported as a key. A word after the `:` of a member and a space, as in `msg: Error: file not found` or `url: https://example.com`, is that member's value instead, an unquoted string. Without the space, as in `a:b: 1`, the first `:` was most likely meant as part of the key.
|
|
729
|
+
const spacesStart = skipSpacesBack(this.#source, this.#index);
|
|
730
|
+
const isMemberValue = spacesStart < this.#index && this.#code(spacesStart - 1) === COLON;
|
|
731
|
+
if (!isMemberValue && this.#code(wordEnd) === COLON) {
|
|
732
|
+
this.#fail(`Expected a value, but found the key ${abbreviate(word)}`);
|
|
733
|
+
}
|
|
734
|
+
this.#diagnoseTableHeader(true);
|
|
735
|
+
this.#fail(describeUnknownWord(word, this.#unquotedText()));
|
|
736
|
+
}
|
|
737
|
+
this.#fail(`Expected a value, but found ${this.#describeHere()}${this.#isBlockScalarIndicator() ? '. Write a multiline string as a block string, between \'\'\' lines' : this.#slashCommentHint()}`);
|
|
738
|
+
}
|
|
739
|
+
/*
|
|
740
|
+
The text from `start` that was most likely meant as one unquoted string, such as `John Smith`: up to the end of the line, a comma, a closing bracket, or a comment after a space or a tab, without the spaces and tabs at its end.
|
|
741
|
+
*/
|
|
742
|
+
#unquotedText(start = this.#index) {
|
|
743
|
+
const source = this.#source;
|
|
744
|
+
const limit = Math.min(source.length, start + MAX_DIAGNOSED_LENGTH);
|
|
745
|
+
let end = start;
|
|
746
|
+
while (end < limit && !isUnquotedTextEnd(source, end)) {
|
|
747
|
+
end++;
|
|
748
|
+
}
|
|
749
|
+
return source.slice(start, skipSpacesBack(source, end));
|
|
750
|
+
}
|
|
751
|
+
/*
|
|
752
|
+
A TOML table header, as in `[server]` or `[[servers]]` on a line of its own, reads as an array that holds a word, or as a key that starts with "[". Where a value is expected, `isValue`, it is only a table header when its bracket opens the document, because a table cannot be written as an item of an array inside it.
|
|
753
|
+
*/
|
|
754
|
+
#diagnoseTableHeader(isValue = false) {
|
|
755
|
+
const source = this.#source;
|
|
756
|
+
const lineStart = source.lastIndexOf('\n', this.#index - 1) + 1;
|
|
757
|
+
if (isValue && skipSpaces(source, lineStart) !== this.#documentStart) {
|
|
758
|
+
return;
|
|
759
|
+
}
|
|
760
|
+
const line = source.slice(lineStart, Math.min(findLineEnd(source, this.#index), lineStart + MAX_DIAGNOSED_LENGTH));
|
|
761
|
+
// The name is a valid key, so that the suggestions are valid: a dot has a segment on each side.
|
|
762
|
+
const match = /^[\t ]*\[\[?(?<name>[A-Z_a-z][\w\-]*(?:\.[\w\-]+)*)\]\]?[\t ]*$/v.exec(line);
|
|
763
|
+
if (match === null) {
|
|
764
|
+
return;
|
|
765
|
+
}
|
|
766
|
+
const { name } = match.groups;
|
|
767
|
+
this.#fail(`There are no table headers. Write the table as an object, as in ${abbreviate(name)}: {…}, or with dotted keys, as in ${abbreviate(name)}.key: …`);
|
|
768
|
+
}
|
|
769
|
+
/*
|
|
770
|
+
A YAML literal block scalar indicator, as in `key: |` or `key: |-`, at the end of its line.
|
|
771
|
+
*/
|
|
772
|
+
#isBlockScalarIndicator() {
|
|
773
|
+
const code = this.#code();
|
|
774
|
+
// A folded block scalar, `>`, joins its lines, which a block string does not, so only `|` gets the hint.
|
|
775
|
+
if (code !== 0x7C /* | */) {
|
|
776
|
+
return false;
|
|
777
|
+
}
|
|
778
|
+
const chomping = this.#code(this.#index + 1);
|
|
779
|
+
const end = skipSpaces(this.#source, chomping === DASH || chomping === 0x2B /* + */ ? this.#index + 2 : this.#index + 1);
|
|
780
|
+
const next = this.#code(end);
|
|
781
|
+
return next === LF || Number.isNaN(next);
|
|
782
|
+
}
|
|
783
|
+
/*
|
|
784
|
+
The hint for a `//` comment, as in JavaScript.
|
|
785
|
+
*/
|
|
786
|
+
#slashCommentHint() {
|
|
787
|
+
return this.#code() === SLASH && this.#code(this.#index + 1) === SLASH ? '. A comment starts with “#”' : '';
|
|
788
|
+
}
|
|
789
|
+
#describeHere() {
|
|
790
|
+
if (this.#isAtEnd()) {
|
|
791
|
+
return 'the end of the document';
|
|
792
|
+
}
|
|
793
|
+
const code = this.#source.codePointAt(this.#index);
|
|
794
|
+
return code === LF ? 'a line break' : describeCharacter(code);
|
|
795
|
+
}
|
|
796
|
+
#parseString() {
|
|
797
|
+
const source = this.#source;
|
|
798
|
+
const code = this.#code();
|
|
799
|
+
if (source.charCodeAt(this.#index + 1) === code && source.charCodeAt(this.#index + 2) === code) {
|
|
800
|
+
return this.#parseBlockString(code);
|
|
801
|
+
}
|
|
802
|
+
return code === SINGLE_QUOTE ? this.#parseLiteralString() : this.#parseEscapedString();
|
|
803
|
+
}
|
|
804
|
+
/*
|
|
805
|
+
`'...'` has no escapes, so its value is the source text between the quotes.
|
|
806
|
+
*/
|
|
807
|
+
#parseLiteralString() {
|
|
808
|
+
const source = this.#source;
|
|
809
|
+
const start = this.#index;
|
|
810
|
+
// Stops at the first quote or line break, so that a one-line document with many strings stays linear.
|
|
811
|
+
LITERAL_STRING_END.lastIndex = start + 1;
|
|
812
|
+
const match = LITERAL_STRING_END.exec(source);
|
|
813
|
+
if (match === null || source.charCodeAt(match.index) === LF) {
|
|
814
|
+
this.#fail('Unterminated string. A \'...\' string must end on the line it starts on; use a block string (\'\'\') for multiple lines', start);
|
|
815
|
+
}
|
|
816
|
+
const end = match.index;
|
|
817
|
+
this.#index = end + 1;
|
|
818
|
+
return source.slice(start + 1, end);
|
|
819
|
+
}
|
|
820
|
+
#parseEscapedString() {
|
|
821
|
+
const source = this.#source;
|
|
822
|
+
const start = this.#index;
|
|
823
|
+
let chunkStart = start + 1;
|
|
824
|
+
let value = '';
|
|
825
|
+
for (;;) {
|
|
826
|
+
ESCAPED_STRING_SPECIAL.lastIndex = chunkStart;
|
|
827
|
+
const match = ESCAPED_STRING_SPECIAL.exec(source);
|
|
828
|
+
if (match === null || match[0] === '\n') {
|
|
829
|
+
this.#fail('Unterminated string. A "..." string must end on the line it starts on; use a block string (""") for multiple lines', start);
|
|
830
|
+
}
|
|
831
|
+
value += source.slice(chunkStart, match.index);
|
|
832
|
+
if (match[0] === '"') {
|
|
833
|
+
this.#index = match.index + 1;
|
|
834
|
+
return value;
|
|
835
|
+
}
|
|
836
|
+
const { text, end } = this.#parseEscape(match.index);
|
|
837
|
+
value += text;
|
|
838
|
+
chunkStart = end;
|
|
839
|
+
}
|
|
840
|
+
}
|
|
841
|
+
/*
|
|
842
|
+
Decodes the escape at `offset`, which is a backslash.
|
|
843
|
+
*/
|
|
844
|
+
#parseEscape(offset) {
|
|
845
|
+
const source = this.#source;
|
|
846
|
+
const character = source[offset + 1];
|
|
847
|
+
const simple = SIMPLE_ESCAPES.get(character ?? '');
|
|
848
|
+
if (simple !== undefined) {
|
|
849
|
+
return { text: simple, end: offset + 2 };
|
|
850
|
+
}
|
|
851
|
+
if (character === 'u') {
|
|
852
|
+
UNICODE_ESCAPE.lastIndex = offset + 1;
|
|
853
|
+
const match = UNICODE_ESCAPE.exec(source);
|
|
854
|
+
if (match === null) {
|
|
855
|
+
this.#fail(describeBadUnicodeEscape(source, offset + 1), offset);
|
|
856
|
+
}
|
|
857
|
+
const { hex } = match.groups;
|
|
858
|
+
const codePoint = Number.parseInt(hex, 16);
|
|
859
|
+
// The value is checked before the spelling, so that the leading zeros error never suggests an escape that is not allowed either.
|
|
860
|
+
if (codePoint === 0x0D) {
|
|
861
|
+
this.#fail(String.raw `A carriage return (U+000D) cannot be represented, so \u{${hex}} is not allowed`, offset);
|
|
862
|
+
}
|
|
863
|
+
if (codePoint >= 0xD8_00 && codePoint <= 0xDF_FF) {
|
|
864
|
+
this.#fail(String.raw `\u{${hex}} is a surrogate, which is not a Unicode scalar value`, offset);
|
|
865
|
+
}
|
|
866
|
+
if (codePoint > 0x10_FF_FF) {
|
|
867
|
+
this.#fail(String.raw `\u{${hex}} is above U+10FFFF, the largest Unicode scalar value`, offset);
|
|
868
|
+
}
|
|
869
|
+
if (hex.length > 1 && hex.startsWith('0')) {
|
|
870
|
+
this.#fail(String.raw `A Unicode escape may not have leading zeros; write \u{${hex.replace(/^0+(?=.)/v, '')}}`, offset);
|
|
871
|
+
}
|
|
872
|
+
return { text: String.fromCodePoint(codePoint), end: offset + 1 + match[0].length };
|
|
873
|
+
}
|
|
874
|
+
// A backslash at the end of the document is the same mistake as one at the end of a line.
|
|
875
|
+
const reason = ESCAPE_MISTAKES.get(character ?? '\n') ?? `Unknown escape “\\${String.fromCodePoint(source.codePointAt(offset + 1))}”. The escapes are \\\\, \\", \\n, \\t, and \\u{…}; use a '...' string for literal backslashes`;
|
|
876
|
+
this.#fail(reason, offset);
|
|
877
|
+
}
|
|
878
|
+
/*
|
|
879
|
+
A block string opens with a run of three or more quotes and a line break, and closes at the first line whose first non-whitespace content is a run of exactly that many quotes. The closing line's indentation is removed from every content line, except a blank one, which holds only spaces and tabs and becomes an empty line.
|
|
880
|
+
*/
|
|
881
|
+
#parseBlockString(quote) {
|
|
882
|
+
const source = this.#source;
|
|
883
|
+
const start = this.#index;
|
|
884
|
+
let delimiterLength = 0;
|
|
885
|
+
while (source.charCodeAt(start + delimiterLength) === quote) {
|
|
886
|
+
delimiterLength++;
|
|
887
|
+
}
|
|
888
|
+
this.#index = start + delimiterLength;
|
|
889
|
+
if (this.#isAtEnd()) {
|
|
890
|
+
this.#fail('Unterminated block string', start);
|
|
891
|
+
}
|
|
892
|
+
// As in Swift, the opening delimiter is followed directly by a line break, not even by trailing whitespace.
|
|
893
|
+
if (this.#code() !== LF) {
|
|
894
|
+
this.#fail('A block string\'s opening delimiter must be followed directly by a line break, and its content starts on the next line');
|
|
895
|
+
}
|
|
896
|
+
const contentStart = this.#index + 1;
|
|
897
|
+
const closing = findBlockStringEnd(source, contentStart, quote, delimiterLength);
|
|
898
|
+
if (closing === undefined) {
|
|
899
|
+
this.#fail(`Unterminated block string${this.#describeInlineClosingDelimiter(contentStart, quote, delimiterLength)}`, start);
|
|
900
|
+
}
|
|
901
|
+
const indentation = source.slice(closing.lineStart, closing.delimiterStart);
|
|
902
|
+
this.#index = closing.delimiterStart + delimiterLength;
|
|
903
|
+
const contents = [];
|
|
904
|
+
for (let lineStart = contentStart; lineStart < closing.lineStart; lineStart = findLineEnd(source, lineStart) + 1) {
|
|
905
|
+
const line = source.slice(lineStart, findLineEnd(source, lineStart));
|
|
906
|
+
// A blank line, which is empty or holds only spaces and tabs, may leave out the indentation, and it becomes an empty line. Swift is stricter here: there only a completely empty line may leave it out.
|
|
907
|
+
if (isBlankLine(line)) {
|
|
908
|
+
contents.push({ text: '', offset: lineStart });
|
|
909
|
+
}
|
|
910
|
+
else if (line.startsWith(indentation)) {
|
|
911
|
+
contents.push({ text: line.slice(indentation.length), offset: lineStart + indentation.length });
|
|
912
|
+
}
|
|
913
|
+
else {
|
|
914
|
+
this.#fail('This line does not start with the indentation of its block string\'s closing delimiter. Every line except a blank one must start with exactly the same spaces and tabs', lineStart);
|
|
915
|
+
}
|
|
916
|
+
}
|
|
917
|
+
// Blank lines directly after the opening delimiter and directly before the closing one are not content. Every blank line is empty by now.
|
|
918
|
+
let first = 0;
|
|
919
|
+
let last = contents.length;
|
|
920
|
+
while (first < last && contents[first].text === '') {
|
|
921
|
+
first++;
|
|
922
|
+
}
|
|
923
|
+
while (last > first && contents[last - 1].text === '') {
|
|
924
|
+
last--;
|
|
925
|
+
}
|
|
926
|
+
const kept = contents.slice(first, last);
|
|
927
|
+
return quote === SINGLE_QUOTE ? kept.map(line => line.text).join('\n') : kept.map(line => this.#unescapeLine(line.text, line.offset)).join('\n');
|
|
928
|
+
}
|
|
929
|
+
/*
|
|
930
|
+
The hint for a content line that ends with the delimiter, as TOML allows. Adding a closing line after it would keep the delimiter as content.
|
|
931
|
+
*/
|
|
932
|
+
#describeInlineClosingDelimiter(contentStart, quote, delimiterLength) {
|
|
933
|
+
const source = this.#source;
|
|
934
|
+
const quoteCharacter = String.fromCharCode(quote);
|
|
935
|
+
const delimiter = quoteCharacter.repeat(delimiterLength);
|
|
936
|
+
let line = source.slice(0, contentStart).split('\n').length;
|
|
937
|
+
for (let lineStart = contentStart; lineStart < source.length; lineStart = findLineEnd(source, lineStart) + 1) {
|
|
938
|
+
const text = source.slice(skipSpaces(source, lineStart), skipSpacesBack(source, findLineEnd(source, lineStart)));
|
|
939
|
+
// A line of only quotes, longer than the delimiter, is content.
|
|
940
|
+
if (text.endsWith(delimiter) && text !== quoteCharacter.repeat(text.length)) {
|
|
941
|
+
return `. Its closing delimiter must start a line, so move the ${delimiter} at the end of line ${line} to a new line`;
|
|
942
|
+
}
|
|
943
|
+
line++;
|
|
944
|
+
}
|
|
945
|
+
return '';
|
|
946
|
+
}
|
|
947
|
+
#unescapeLine(text, offset) {
|
|
948
|
+
let value = '';
|
|
949
|
+
let chunkStart = 0;
|
|
950
|
+
for (let index = text.indexOf('\\'); index !== -1; index = text.indexOf('\\', chunkStart)) {
|
|
951
|
+
value += text.slice(chunkStart, index);
|
|
952
|
+
const escape = this.#parseEscape(offset + index);
|
|
953
|
+
value += escape.text;
|
|
954
|
+
chunkStart = escape.end - offset;
|
|
955
|
+
}
|
|
956
|
+
return value + text.slice(chunkStart);
|
|
957
|
+
}
|
|
958
|
+
#parseNumberOrInstant() {
|
|
959
|
+
const plain = this.#parsePlainNumber();
|
|
960
|
+
if (plain !== undefined) {
|
|
961
|
+
return plain;
|
|
962
|
+
}
|
|
963
|
+
const start = this.#index;
|
|
964
|
+
const end = findNumberEnd(this.#source, start);
|
|
965
|
+
const text = this.#source.slice(start, end);
|
|
966
|
+
this.#index = end;
|
|
967
|
+
if (INSTANT_PREFIX.test(text)) {
|
|
968
|
+
return this.#parseInstant(text, start);
|
|
969
|
+
}
|
|
970
|
+
const kind = classifyNumber(text);
|
|
971
|
+
if (kind === undefined && isDurationLike(text)) {
|
|
972
|
+
return this.#parseDuration(text, start);
|
|
973
|
+
}
|
|
974
|
+
switch (kind) {
|
|
975
|
+
case 'int': {
|
|
976
|
+
if (text === '-0') {
|
|
977
|
+
this.#fail('“-0” is not allowed, because zero has one spelling: 0', start);
|
|
978
|
+
}
|
|
979
|
+
return this.#integer(text.replaceAll('_', ''), text, start);
|
|
980
|
+
}
|
|
981
|
+
case 'float': {
|
|
982
|
+
const value = Number(text.replaceAll('_', ''));
|
|
983
|
+
if (!Number.isFinite(value)) {
|
|
984
|
+
this.#fail(`${abbreviate(text)} is too large to be a finite float. Use ${value < 0 ? '-infinity' : 'infinity'} if you mean it`, start);
|
|
985
|
+
}
|
|
986
|
+
if (value === 0) {
|
|
987
|
+
// A nonzero digit before the exponent means the literal is not zero, so it underflowed.
|
|
988
|
+
if (/[1-9]/v.test(text.split('e', 1)[0])) {
|
|
989
|
+
this.#fail(`${abbreviate(text)} is too small to be told apart from zero. Write 0.0 if you mean zero`, start);
|
|
990
|
+
}
|
|
991
|
+
// Negative zero is the same value as zero.
|
|
992
|
+
return 0;
|
|
993
|
+
}
|
|
994
|
+
return value;
|
|
995
|
+
}
|
|
996
|
+
default: {
|
|
997
|
+
break;
|
|
998
|
+
}
|
|
999
|
+
}
|
|
1000
|
+
this.#fail(this.#describeThousandsSeparator(text, start) ?? describeBadNumber(text, this.#unquotedText(start)), start);
|
|
1001
|
+
}
|
|
1002
|
+
/*
|
|
1003
|
+
A comma used as a thousands separator, as in `[1,000]`, makes the group after it a separate item, which fails only when it has a leading zero. Writing that item in octal, as the leading zero message suggests, would parse to the wrong items.
|
|
1004
|
+
*/
|
|
1005
|
+
#describeThousandsSeparator(text, start) {
|
|
1006
|
+
if (!/^0\d{2}$/v.test(text) || this.#code(start - 1) !== COMMA) {
|
|
1007
|
+
return undefined;
|
|
1008
|
+
}
|
|
1009
|
+
const getGroupStart = (groupEnd) => {
|
|
1010
|
+
let groupStart = groupEnd;
|
|
1011
|
+
while (isDigit(this.#code(groupStart - 1))) {
|
|
1012
|
+
groupStart--;
|
|
1013
|
+
}
|
|
1014
|
+
return groupStart;
|
|
1015
|
+
};
|
|
1016
|
+
let groupEnd = start - 1;
|
|
1017
|
+
let numberStart = getGroupStart(groupEnd);
|
|
1018
|
+
// Earlier groups of three digits, as in `12,345,000`.
|
|
1019
|
+
while (groupEnd - numberStart === 3 && this.#code(numberStart - 1) === COMMA && isDigit(this.#code(numberStart - 2))) {
|
|
1020
|
+
groupEnd = numberStart - 1;
|
|
1021
|
+
numberStart = getGroupStart(groupEnd);
|
|
1022
|
+
}
|
|
1023
|
+
const before = this.#code(numberStart - 1);
|
|
1024
|
+
// The first group has one to three digits, which are not the end of a float, of a number with underscores, or of a number with a radix prefix.
|
|
1025
|
+
if (numberStart === groupEnd
|
|
1026
|
+
|| before === DOT
|
|
1027
|
+
|| before === 0x5F /* _ */
|
|
1028
|
+
|| numberStart < groupEnd - 3
|
|
1029
|
+
|| isLetter(before)) {
|
|
1030
|
+
return undefined;
|
|
1031
|
+
}
|
|
1032
|
+
// The sign belongs to the number, as in `-1,000`.
|
|
1033
|
+
if (before === DASH) {
|
|
1034
|
+
numberStart--;
|
|
1035
|
+
}
|
|
1036
|
+
const groupsStart = start + text.length;
|
|
1037
|
+
const groups = /^(?:,\d{3}(?!\d))*/v.exec(this.#source.slice(groupsStart, groupsStart + MAX_DIAGNOSED_LENGTH))[0];
|
|
1038
|
+
const number = this.#source.slice(numberStart, groupsStart + groups.length);
|
|
1039
|
+
const reason = `Leading zeros are not allowed in a decimal number. A comma separates items, so ${abbreviate(number)} is not one number`;
|
|
1040
|
+
// Both spellings are the same int, so they are suggested only when it is in range.
|
|
1041
|
+
const withUnderscores = number.replaceAll(',', '_');
|
|
1042
|
+
return isValidValue(withUnderscores) ? `${reason}. Write it as ${abbreviate(withUnderscores)} or ${abbreviate(number.replaceAll(',', ''))}` : reason;
|
|
1043
|
+
}
|
|
1044
|
+
/*
|
|
1045
|
+
The common case, a short decimal int or float such as `8080`, `-3`, or `30.5`, without the general path's maximal-run scan and classification. Anything else, including every error, returns `undefined` and is left to the general path.
|
|
1046
|
+
*/
|
|
1047
|
+
#parsePlainNumber() {
|
|
1048
|
+
const source = this.#source;
|
|
1049
|
+
const start = this.#index;
|
|
1050
|
+
let index = source.charCodeAt(start) === DASH ? start + 1 : start;
|
|
1051
|
+
const integerStart = index;
|
|
1052
|
+
while (isDigit(source.charCodeAt(index))) {
|
|
1053
|
+
index++;
|
|
1054
|
+
}
|
|
1055
|
+
const integerLength = index - integerStart;
|
|
1056
|
+
// No digits, or a leading zero followed by more digits, which the general path reports as an error or reads as an instant before the year 1000.
|
|
1057
|
+
if (integerLength === 0 || (integerLength > 1 && source.charCodeAt(integerStart) === 0x30)) {
|
|
1058
|
+
return;
|
|
1059
|
+
}
|
|
1060
|
+
let isFloat = false;
|
|
1061
|
+
if (source.charCodeAt(index) === DOT) {
|
|
1062
|
+
const fractionStart = index + 1;
|
|
1063
|
+
index = fractionStart;
|
|
1064
|
+
while (isDigit(source.charCodeAt(index))) {
|
|
1065
|
+
index++;
|
|
1066
|
+
}
|
|
1067
|
+
if (index === fractionStart) {
|
|
1068
|
+
return;
|
|
1069
|
+
}
|
|
1070
|
+
isFloat = true;
|
|
1071
|
+
}
|
|
1072
|
+
// At most 15 digits, so an int is exact as a number. A character that may continue a number, such as `e`, `_`, or `-`, needs the general path.
|
|
1073
|
+
if (index - integerStart > 15 || !isValueEnd(source.charCodeAt(index))) {
|
|
1074
|
+
return;
|
|
1075
|
+
}
|
|
1076
|
+
const value = Number(source.slice(start, index));
|
|
1077
|
+
if (isFloat) {
|
|
1078
|
+
this.#index = index;
|
|
1079
|
+
// Negative zero is the same value as zero.
|
|
1080
|
+
return value === 0 ? 0 : value;
|
|
1081
|
+
}
|
|
1082
|
+
// `-0` is an error, which the general path reports.
|
|
1083
|
+
if (value === 0 && index - start > 1) {
|
|
1084
|
+
return;
|
|
1085
|
+
}
|
|
1086
|
+
this.#index = index;
|
|
1087
|
+
return this.#integers === 'bigint' ? BigInt(value) : value;
|
|
1088
|
+
}
|
|
1089
|
+
/*
|
|
1090
|
+
`digits` is the int without underscores. More than 64 significant digits is out of range in any radix, which is decided before `BigInt` spends time on a huge digit string.
|
|
1091
|
+
*/
|
|
1092
|
+
#integer(digits, text, start) {
|
|
1093
|
+
const isRadix = digits.length > 1 && digits.charCodeAt(1) > 0x39;
|
|
1094
|
+
let significant = digits.charCodeAt(0) === DASH ? 1 : 0;
|
|
1095
|
+
if (isRadix) {
|
|
1096
|
+
significant = 2;
|
|
1097
|
+
while (digits.charCodeAt(significant) === 0x30) {
|
|
1098
|
+
significant++;
|
|
1099
|
+
}
|
|
1100
|
+
}
|
|
1101
|
+
const value = digits.length - significant > 64 ? undefined : BigInt(isRadix ? `${digits.slice(0, 2)}${digits.length === significant ? '0' : digits.slice(significant)}` : digits);
|
|
1102
|
+
if (value === undefined || value < INT64_MIN || value > INT64_MAX) {
|
|
1103
|
+
this.#fail(`The integer ${abbreviate(text)} is outside the 64-bit range (-9223372036854775808 to 9223372036854775807)`, start);
|
|
1104
|
+
}
|
|
1105
|
+
if (this.#integers === 'bigint') {
|
|
1106
|
+
return value;
|
|
1107
|
+
}
|
|
1108
|
+
if (value < Number.MIN_SAFE_INTEGER || value > Number.MAX_SAFE_INTEGER) {
|
|
1109
|
+
this.#fail(`The integer ${abbreviate(text)} cannot be represented exactly as a JavaScript number. Remove the \`integers: 'number'\` option to get a BigInt`, start);
|
|
1110
|
+
}
|
|
1111
|
+
return Number(value);
|
|
1112
|
+
}
|
|
1113
|
+
#parseInstant(text, start) {
|
|
1114
|
+
const match = text.length > MAX_DIAGNOSED_LENGTH ? null : INSTANT.exec(text);
|
|
1115
|
+
if (match === null) {
|
|
1116
|
+
// A date, a space, and a time, as TOML and Python write a date and time.
|
|
1117
|
+
const timeStart = this.#index + 1;
|
|
1118
|
+
const end = this.#code() === SPACE && isDigit(this.#code(timeStart)) ? findNumberEnd(this.#source, timeStart) : this.#index;
|
|
1119
|
+
const time = this.#source.slice(timeStart, end);
|
|
1120
|
+
// More of the instant may follow, as in `14:00:00,5Z` or `14:00:00[Europe/Oslo]`, so it may have an offset.
|
|
1121
|
+
const next = this.#code(end);
|
|
1122
|
+
const isWholeValue = isValueEnd(next) && !(next === COMMA && isDigit(this.#code(end + 1)));
|
|
1123
|
+
this.#fail(describeBadInstant(text, time, isWholeValue), start);
|
|
1124
|
+
}
|
|
1125
|
+
const { year, month, day, hour, minute, second, fraction, offset, offsetHour, offsetMinute } = match.groups;
|
|
1126
|
+
const check = (isValid, reason) => {
|
|
1127
|
+
if (!isValid) {
|
|
1128
|
+
this.#fail(`Invalid instant ${abbreviate(text)}: ${reason}`, start);
|
|
1129
|
+
}
|
|
1130
|
+
};
|
|
1131
|
+
check(fraction === undefined || fraction.length <= 9, 'a fractional second has at most nine digits');
|
|
1132
|
+
check(year !== '0000', 'the year must be 0001 to 9999');
|
|
1133
|
+
check(month >= '01' && month <= '12', 'the month must be 01 to 12');
|
|
1134
|
+
const lastDay = daysInMonth(Number(year), Number(month));
|
|
1135
|
+
check(day >= '01' && Number(day) <= lastDay, `the day must be 01 to ${lastDay} in that month`);
|
|
1136
|
+
check(hour <= '23', 'the hour must be 00 to 23');
|
|
1137
|
+
check(minute <= '59', 'the minute must be 00 to 59');
|
|
1138
|
+
check(second <= '59', 'the second must be 00 to 59, and a leap second is not representable');
|
|
1139
|
+
if (offset !== 'Z') {
|
|
1140
|
+
check(offsetHour <= '23', 'the offset hour must be 00 to 23');
|
|
1141
|
+
check(offsetMinute <= '59', 'the offset minute must be 00 to 59');
|
|
1142
|
+
check(offset !== '-00:00', '-00:00 means “offset unknown” in RFC 3339, which is not representable; use Z or +00:00');
|
|
1143
|
+
}
|
|
1144
|
+
// `Date.parse()` reads this format exactly, for every year from 0001 to 9999 and every offset, so the range check needs no `Temporal`.
|
|
1145
|
+
const milliseconds = Date.parse(`${year}-${month}-${day}T${hour}:${minute}:${second}${offset}`);
|
|
1146
|
+
const nanoseconds = (BigInt(milliseconds) * 1000000n) + BigInt((fraction ?? '').padEnd(9, '0'));
|
|
1147
|
+
check(nanoseconds >= MIN_INSTANT && nanoseconds <= MAX_INSTANT, 'in UTC it falls outside the years 0001 to 9999');
|
|
1148
|
+
return this.#createTime(new Time('Instant', nanoseconds));
|
|
1149
|
+
}
|
|
1150
|
+
/*
|
|
1151
|
+
A scan of the parts in order, so that each error names the part it is about.
|
|
1152
|
+
*/
|
|
1153
|
+
#parseDuration(text, start) {
|
|
1154
|
+
const fail = (reason) => {
|
|
1155
|
+
this.#fail(`Invalid duration ${abbreviate(text)}: ${reason}`, start);
|
|
1156
|
+
};
|
|
1157
|
+
const isNegative = text.charCodeAt(0) === DASH;
|
|
1158
|
+
let index = isNegative ? 1 : 0;
|
|
1159
|
+
let previousRank = -1;
|
|
1160
|
+
let total = 0n;
|
|
1161
|
+
// The fraction of the last part, and that part's unit size. Only the last part may have one, so the value checks wait until every part is read.
|
|
1162
|
+
let fraction = '';
|
|
1163
|
+
let size = 0n;
|
|
1164
|
+
while (index < text.length) {
|
|
1165
|
+
const digitsEnd = skipIntegerPart(text, index);
|
|
1166
|
+
const next = text.charCodeAt(digitsEnd);
|
|
1167
|
+
if (digitsEnd === index) {
|
|
1168
|
+
const code = text.charCodeAt(index);
|
|
1169
|
+
if (code === DASH && isDigit(text.charCodeAt(index + 1))) {
|
|
1170
|
+
fail('only the whole duration takes a sign, as in -1h30m');
|
|
1171
|
+
}
|
|
1172
|
+
fail(code === 0x2B /* + */ ? 'a “+” sign is not allowed' : `expected a number at “${abbreviate(text.slice(index), 10)}”`);
|
|
1173
|
+
}
|
|
1174
|
+
if (text.charCodeAt(index) === 0x30 && (next === 0x5F /* _ */ || isDigit(next))) {
|
|
1175
|
+
fail('leading zeros are not allowed');
|
|
1176
|
+
}
|
|
1177
|
+
if (next === 0x5F /* _ */) {
|
|
1178
|
+
fail('an underscore must be between two digits');
|
|
1179
|
+
}
|
|
1180
|
+
fraction = '';
|
|
1181
|
+
let unitStart = digitsEnd;
|
|
1182
|
+
if (next === DOT) {
|
|
1183
|
+
unitStart = skipDigits(text, digitsEnd + 1, isDigit);
|
|
1184
|
+
if (unitStart === digitsEnd + 1) {
|
|
1185
|
+
fail('a “.” must be followed by a digit');
|
|
1186
|
+
}
|
|
1187
|
+
if (text.charCodeAt(unitStart) === 0x5F /* _ */) {
|
|
1188
|
+
fail('an underscore must be between two digits');
|
|
1189
|
+
}
|
|
1190
|
+
// Trailing zeros change nothing, and leaving them out keeps the arithmetic small however many there are.
|
|
1191
|
+
fraction = trimTrailingZeros(text.slice(digitsEnd + 1, unitStart).replaceAll('_', ''));
|
|
1192
|
+
}
|
|
1193
|
+
let unitEnd = unitStart;
|
|
1194
|
+
while (isLetter(text.charCodeAt(unitEnd))) {
|
|
1195
|
+
unitEnd++;
|
|
1196
|
+
}
|
|
1197
|
+
const unit = text.slice(unitStart, unitEnd);
|
|
1198
|
+
const rank = DURATION_UNIT_NAMES.indexOf(unit);
|
|
1199
|
+
if (rank === -1) {
|
|
1200
|
+
fail(describeBadDurationUnit(unit, text));
|
|
1201
|
+
}
|
|
1202
|
+
if (rank <= previousRank) {
|
|
1203
|
+
fail('the units are in the order h, m, s, ms, us, ns, and each appears at most once');
|
|
1204
|
+
}
|
|
1205
|
+
if (next === DOT && isDigit(text.charCodeAt(unitEnd))) {
|
|
1206
|
+
fail('only the last part may have a fraction');
|
|
1207
|
+
}
|
|
1208
|
+
size = DURATION_UNITS.get(unit);
|
|
1209
|
+
const digits = text.slice(index, digitsEnd).replaceAll('_', '');
|
|
1210
|
+
// More than 19 digits is out of range in any unit, which is decided before `BigInt` spends time on a huge digit string.
|
|
1211
|
+
total += digits.length > 19 ? INT64_MAX * 2n : BigInt(digits) * size;
|
|
1212
|
+
index = unitEnd;
|
|
1213
|
+
previousRank = rank;
|
|
1214
|
+
}
|
|
1215
|
+
// More than 13 fraction digits without a trailing zero is finer than a nanosecond in any unit, which is decided before `BigInt` spends time on a huge digit string.
|
|
1216
|
+
if (fraction.length > 13) {
|
|
1217
|
+
fail('it is not a whole number of nanoseconds');
|
|
1218
|
+
}
|
|
1219
|
+
if (fraction !== '') {
|
|
1220
|
+
const scale = 10n ** BigInt(fraction.length);
|
|
1221
|
+
const fractionNanoseconds = BigInt(fraction) * size;
|
|
1222
|
+
if (fractionNanoseconds % scale !== 0n) {
|
|
1223
|
+
fail('it is not a whole number of nanoseconds');
|
|
1224
|
+
}
|
|
1225
|
+
total += fractionNanoseconds / scale;
|
|
1226
|
+
}
|
|
1227
|
+
if (isNegative && total === 0n) {
|
|
1228
|
+
fail('“-” is not allowed before zero, because zero has one spelling: 0s');
|
|
1229
|
+
}
|
|
1230
|
+
if (total > (isNegative ? -INT64_MIN : INT64_MAX)) {
|
|
1231
|
+
fail('it is outside the 64-bit range of nanoseconds, about 292 years either way');
|
|
1232
|
+
}
|
|
1233
|
+
return this.#createTime(new Time('Duration', isNegative ? -total : total));
|
|
1234
|
+
}
|
|
1235
|
+
parseDocument() {
|
|
1236
|
+
this.#skipTrivia();
|
|
1237
|
+
this.#documentStart = this.#index;
|
|
1238
|
+
if (this.#isAtEnd()) {
|
|
1239
|
+
this.#fail('A document must contain an object or an array, but this one is empty');
|
|
1240
|
+
}
|
|
1241
|
+
const code = this.#code();
|
|
1242
|
+
let value;
|
|
1243
|
+
if (code === OPEN_BRACE) {
|
|
1244
|
+
value = this.#parseObject(1);
|
|
1245
|
+
}
|
|
1246
|
+
else if (code === OPEN_BRACKET) {
|
|
1247
|
+
value = this.#parseArray(1);
|
|
1248
|
+
}
|
|
1249
|
+
else {
|
|
1250
|
+
value = this.#parseBareObject();
|
|
1251
|
+
}
|
|
1252
|
+
this.#skipTrivia();
|
|
1253
|
+
if (!this.#isAtEnd()) {
|
|
1254
|
+
this.#fail(`Unexpected ${this.#describeHere()} after the end of the document`);
|
|
1255
|
+
}
|
|
1256
|
+
return value;
|
|
1257
|
+
}
|
|
1258
|
+
}
|
|
1259
|
+
/*
|
|
1260
|
+
Whether `text` is one valid value, so that an error message only suggests a fix that works.
|
|
1261
|
+
*/
|
|
1262
|
+
function isValidValue(text) {
|
|
1263
|
+
try {
|
|
1264
|
+
return new Parser(`[${text}]`, 'bigint', time => time).parseDocument().length === 1;
|
|
1265
|
+
}
|
|
1266
|
+
catch {
|
|
1267
|
+
return false;
|
|
1268
|
+
}
|
|
1269
|
+
}
|
|
1270
|
+
/*
|
|
1271
|
+
Whether `text` is an int, in any radix, or a float, following the grammar. A run of digits may have single underscores between digits.
|
|
1272
|
+
*/
|
|
1273
|
+
function classifyNumber(text) {
|
|
1274
|
+
const radix = text.charCodeAt(0) === 0x30 ? RADIX_DIGIT.get(text.charAt(1)) : undefined;
|
|
1275
|
+
if (radix !== undefined) {
|
|
1276
|
+
const end = skipDigits(text, 2, radix);
|
|
1277
|
+
return end > 2 && end === text.length ? 'int' : undefined;
|
|
1278
|
+
}
|
|
1279
|
+
let index = text.charCodeAt(0) === DASH ? 1 : 0;
|
|
1280
|
+
// The integer part is `0` or has no leading zero.
|
|
1281
|
+
const integerEnd = skipIntegerPart(text, index);
|
|
1282
|
+
if (integerEnd === index) {
|
|
1283
|
+
return undefined;
|
|
1284
|
+
}
|
|
1285
|
+
index = integerEnd;
|
|
1286
|
+
let kind = 'int';
|
|
1287
|
+
if (text.charCodeAt(index) === DOT) {
|
|
1288
|
+
const end = skipDigits(text, index + 1, isDigit);
|
|
1289
|
+
if (end === index + 1) {
|
|
1290
|
+
return undefined;
|
|
1291
|
+
}
|
|
1292
|
+
index = end;
|
|
1293
|
+
kind = 'float';
|
|
1294
|
+
}
|
|
1295
|
+
if (text.charCodeAt(index) === 0x65 /* e */) {
|
|
1296
|
+
const isNegative = text.charCodeAt(index + 1) === DASH;
|
|
1297
|
+
const exponentStart = index + (isNegative ? 2 : 1);
|
|
1298
|
+
// The exponent follows the same leading-zero rule as the integer part, and its zero has one spelling, `e0`, so `e-0` is not an exponent.
|
|
1299
|
+
const end = isNegative && text.charCodeAt(exponentStart) === 0x30 ? exponentStart : skipIntegerPart(text, exponentStart);
|
|
1300
|
+
if (end === exponentStart) {
|
|
1301
|
+
return undefined;
|
|
1302
|
+
}
|
|
1303
|
+
index = end;
|
|
1304
|
+
kind = 'float';
|
|
1305
|
+
}
|
|
1306
|
+
return index === text.length ? kind : undefined;
|
|
1307
|
+
}
|
|
1308
|
+
/*
|
|
1309
|
+
Returns the index after a run of digits that may have single underscores between them, or `index` when there is no digit there.
|
|
1310
|
+
*/
|
|
1311
|
+
function skipDigits(text, index, isValidDigit) {
|
|
1312
|
+
if (!isValidDigit(text.charCodeAt(index))) {
|
|
1313
|
+
return index;
|
|
1314
|
+
}
|
|
1315
|
+
index++;
|
|
1316
|
+
for (;;) {
|
|
1317
|
+
const code = text.charCodeAt(index);
|
|
1318
|
+
if (isValidDigit(code)) {
|
|
1319
|
+
index++;
|
|
1320
|
+
}
|
|
1321
|
+
else if (code === 0x5F /* _ */ && isValidDigit(text.charCodeAt(index + 1))) {
|
|
1322
|
+
index += 2;
|
|
1323
|
+
}
|
|
1324
|
+
else {
|
|
1325
|
+
return index;
|
|
1326
|
+
}
|
|
1327
|
+
}
|
|
1328
|
+
}
|
|
1329
|
+
/*
|
|
1330
|
+
Returns the index after an integer part, which is `0` or digits without a leading zero, or `index` when there is none. Anything after a leading `0`, such as the `5` in `05`, is left for the caller to reject.
|
|
1331
|
+
*/
|
|
1332
|
+
function skipIntegerPart(text, index) {
|
|
1333
|
+
return text.charCodeAt(index) === 0x30 ? index + 1 : skipDigits(text, index, isDigit);
|
|
1334
|
+
}
|
|
1335
|
+
function isDigit(code) {
|
|
1336
|
+
return code >= 0x30 && code <= 0x39;
|
|
1337
|
+
}
|
|
1338
|
+
function isLetter(code) {
|
|
1339
|
+
return (code >= 0x41 && code <= 0x5A) || (code >= 0x61 && code <= 0x7A);
|
|
1340
|
+
}
|
|
1341
|
+
/*
|
|
1342
|
+
Whether a token that is not a number or an instant was meant as a duration: its first letter could begin a unit, or a day or a week. A radix prefix, an exponent, and other letters, such as the `T` in `20260919T140000Z` or the `x` in `1.5x`, are left to the number errors.
|
|
1343
|
+
*/
|
|
1344
|
+
function isDurationLike(text) {
|
|
1345
|
+
const start = text.charCodeAt(0) === DASH ? 1 : 0;
|
|
1346
|
+
if (!isDigit(text.charCodeAt(start))) {
|
|
1347
|
+
return false;
|
|
1348
|
+
}
|
|
1349
|
+
// A scan rather than a regular expression, because a fraction may have any number of trailing zeros, so the first letter can be millions of characters in.
|
|
1350
|
+
let index = start;
|
|
1351
|
+
while (isDigit(text.charCodeAt(index)) || text.charCodeAt(index) === 0x5F) {
|
|
1352
|
+
index++;
|
|
1353
|
+
}
|
|
1354
|
+
if (text.charCodeAt(index) === DOT) {
|
|
1355
|
+
index++;
|
|
1356
|
+
while (isDigit(text.charCodeAt(index)) || text.charCodeAt(index) === 0x5F) {
|
|
1357
|
+
index++;
|
|
1358
|
+
}
|
|
1359
|
+
}
|
|
1360
|
+
return /^[dhmnsuw]$/iv.test(text.charAt(index));
|
|
1361
|
+
}
|
|
1362
|
+
function describeBadDurationUnit(unit, text) {
|
|
1363
|
+
if (unit === '') {
|
|
1364
|
+
return 'every number needs a unit: h, m, s, ms, us, or ns';
|
|
1365
|
+
}
|
|
1366
|
+
if (/^d(?:ays?)?$/v.test(unit)) {
|
|
1367
|
+
return 'there is no day unit, because a day is not a fixed length. Write 24h for a fixed 24 hours';
|
|
1368
|
+
}
|
|
1369
|
+
if (/^w(?:eeks?)?$/v.test(unit)) {
|
|
1370
|
+
return 'there is no week unit, because a day is not a fixed length. Write 168h for a fixed 168 hours';
|
|
1371
|
+
}
|
|
1372
|
+
if (unit === 'M') {
|
|
1373
|
+
// A number with `M` alone is more often a size, as in `memory: 512M`, and `512m` would read as minutes.
|
|
1374
|
+
const sizeHint = text.length <= MAX_DIAGNOSED_LENGTH && /^\d[\d_]*M$/v.test(text) ? `. A size, such as 512M, is a string: '${abbreviate(text)}'` : '';
|
|
1375
|
+
return `there is no month unit, because a month is not a fixed length${sizeHint}`;
|
|
1376
|
+
}
|
|
1377
|
+
return DURATION_UNITS.has(unit.toLowerCase()) ? `the units are lowercase: ${unit.toLowerCase()}` : `“${abbreviate(unit, 10)}” is not a unit. The units are h, m, s, ms, us, and ns, and a string must be quoted`;
|
|
1378
|
+
}
|
|
1379
|
+
/*
|
|
1380
|
+
The number of spaces and tabs before `offset` on its line, or `undefined` when something else comes before it.
|
|
1381
|
+
*/
|
|
1382
|
+
function getIndentation(source, offset) {
|
|
1383
|
+
const lineStart = source.lastIndexOf('\n', offset - 1) + 1;
|
|
1384
|
+
return skipSpaces(source, lineStart) === offset ? offset - lineStart : undefined;
|
|
1385
|
+
}
|
|
1386
|
+
function isWordsWithSpaces(text) {
|
|
1387
|
+
for (let index = 0; index < text.length; index++) {
|
|
1388
|
+
const code = text.charCodeAt(index);
|
|
1389
|
+
if (!isSpace(code) && !isBareKeyCharacter(code)) {
|
|
1390
|
+
return false;
|
|
1391
|
+
}
|
|
1392
|
+
}
|
|
1393
|
+
return true;
|
|
1394
|
+
}
|
|
1395
|
+
function daysInMonth(year, month) {
|
|
1396
|
+
if (month === 2) {
|
|
1397
|
+
const isLeapYear = year % 4 === 0 && (year % 100 !== 0 || year % 400 === 0);
|
|
1398
|
+
return isLeapYear ? 29 : 28;
|
|
1399
|
+
}
|
|
1400
|
+
return [4, 6, 9, 11].includes(month) ? 30 : 31;
|
|
1401
|
+
}
|
|
1402
|
+
function isUnquotedTextEnd(source, index) {
|
|
1403
|
+
const code = source.charCodeAt(index);
|
|
1404
|
+
const isComment = code === HASH || (code === SLASH && source.charCodeAt(index + 1) === ASTERISK);
|
|
1405
|
+
return code === LF || code === COMMA || code === CLOSE_BRACKET || code === CLOSE_BRACE || (isComment && isSpace(source.charCodeAt(index - 1)));
|
|
1406
|
+
}
|
|
1407
|
+
/*
|
|
1408
|
+
The example in a message that a string must be quoted, which quotes `text` as a `'...'` string. That cannot hold a `'`, so then there is no simple example.
|
|
1409
|
+
*/
|
|
1410
|
+
function quotingExample(text) {
|
|
1411
|
+
return text.includes('\'') ? '' : `, as in '${abbreviate(text)}'`;
|
|
1412
|
+
}
|
|
1413
|
+
/*
|
|
1414
|
+
`text` is the unquoted string that `word` begins, which the suggestion quotes.
|
|
1415
|
+
*/
|
|
1416
|
+
function describeUnknownWord(word, text) {
|
|
1417
|
+
// A keyword hint is only for the word on its own. In `Yes please` or `Nan Goldin`, following it would leave text behind or change the value.
|
|
1418
|
+
const lowercase = text === word ? word.toLowerCase() : '';
|
|
1419
|
+
if (['true', 'false', 'yes', 'no', 'on', 'off'].includes(lowercase)) {
|
|
1420
|
+
return `“${word}” is not a value. Booleans are written true and false, in lowercase`;
|
|
1421
|
+
}
|
|
1422
|
+
if (['null', 'nil', 'none', 'undefined'].includes(lowercase)) {
|
|
1423
|
+
return `“${word}” is not a value. Null is written null, in lowercase`;
|
|
1424
|
+
}
|
|
1425
|
+
if (lowercase === 'nan') {
|
|
1426
|
+
return 'NaN is not representable. Use null for a missing value';
|
|
1427
|
+
}
|
|
1428
|
+
return ['inf', 'infinity'].includes(lowercase) ? `“${word}” is not a value. Infinity is written infinity, in lowercase` : `Unexpected “${abbreviate(word)}”. A string value must be quoted${quotingExample(text)}`;
|
|
1429
|
+
}
|
|
1430
|
+
function describeBadUnicodeEscape(source, offset) {
|
|
1431
|
+
const rest = source.slice(offset, offset + 12);
|
|
1432
|
+
if (/^u[\dA-Fa-f]{4}/v.test(rest)) {
|
|
1433
|
+
return describeFourDigitEscape(rest);
|
|
1434
|
+
}
|
|
1435
|
+
if (/^u\{[\dA-Fa-f]*[A-F]/v.test(rest)) {
|
|
1436
|
+
return 'A Unicode escape uses lowercase hexadecimal digits';
|
|
1437
|
+
}
|
|
1438
|
+
if (rest.startsWith('u{}')) {
|
|
1439
|
+
return 'A Unicode escape needs one to six hexadecimal digits';
|
|
1440
|
+
}
|
|
1441
|
+
return /^u\{[\da-f]{7}/v.test(rest) ? 'A Unicode escape has at most six hexadecimal digits' : String.raw `A Unicode escape is written \u{…} with one to six lowercase hexadecimal digits`;
|
|
1442
|
+
}
|
|
1443
|
+
/*
|
|
1444
|
+
The JSON form `\uXXXX`, from its `u`, with the escape to write instead. JSON writes a character above U+FFFF as two of them, a surrogate pair, which is one `\u{…}` escape here. Here, a lone surrogate and a carriage return have no escape.
|
|
1445
|
+
*/
|
|
1446
|
+
function describeFourDigitEscape(rest) {
|
|
1447
|
+
const code = Number.parseInt(rest.slice(1, 5), 16);
|
|
1448
|
+
const low = /^\\u[\dA-Fa-f]{4}/v.test(rest.slice(5)) ? rest.slice(7, 11) : undefined;
|
|
1449
|
+
const lowCode = low === undefined ? NaN : Number.parseInt(low, 16);
|
|
1450
|
+
if (code >= 0xD8_00 && code <= 0xDB_FF && lowCode >= 0xDC_00 && lowCode <= 0xDF_FF) {
|
|
1451
|
+
const codePoint = 0x1_00_00 + ((code - 0xD8_00) * 0x4_00) + (lowCode - 0xDC_00);
|
|
1452
|
+
return `The four-digit \\${rest.slice(0, 5)}\\u${low} form is not an escape. Write \\u{${codePoint.toString(16)}}`;
|
|
1453
|
+
}
|
|
1454
|
+
const form = `The four-digit \\${rest.slice(0, 5)} form is not an escape`;
|
|
1455
|
+
if (code >= 0xD8_00 && code <= 0xDF_FF) {
|
|
1456
|
+
return String.raw `${form}, and a lone surrogate is not a Unicode scalar value. Write the character it is half of as one \u{…} escape`;
|
|
1457
|
+
}
|
|
1458
|
+
return code === 0x0D ? `${form}, and a carriage return (U+000D) cannot be represented` : String.raw `${form}. Write \u{${code.toString(16)}}`;
|
|
1459
|
+
}
|
|
1460
|
+
/*
|
|
1461
|
+
`unquotedText` is the unquoted string that `fullText` begins, for the suggestion to quote it.
|
|
1462
|
+
*/
|
|
1463
|
+
function describeBadNumber(fullText, unquotedText) {
|
|
1464
|
+
const text = fullText.slice(0, MAX_DIAGNOSED_LENGTH);
|
|
1465
|
+
if (text.includes('+')) {
|
|
1466
|
+
return 'A “+” sign is not allowed in a number, including in an exponent';
|
|
1467
|
+
}
|
|
1468
|
+
if (/^-?0[BOX]/v.test(text)) {
|
|
1469
|
+
return 'A number prefix is lowercase: 0x, 0o, or 0b';
|
|
1470
|
+
}
|
|
1471
|
+
// The whole number, not the part cut at the diagnosed length, because a cut can end before the digit or the underscore that the message is about. Each check below takes linear time.
|
|
1472
|
+
const radixMatch = /^(?<sign>-?)0(?<radix>[box])(?<digits>.*)$/v.exec(fullText);
|
|
1473
|
+
if (radixMatch !== null) {
|
|
1474
|
+
const { sign, radix, digits } = radixMatch.groups;
|
|
1475
|
+
const name = RADIX_NAME[radix];
|
|
1476
|
+
if (sign === '-') {
|
|
1477
|
+
return `${RADIX_ARTICLE[radix]} ${name} integer cannot have a sign, because it states a bit pattern rather than a quantity`;
|
|
1478
|
+
}
|
|
1479
|
+
if (digits === '') {
|
|
1480
|
+
return `Expected ${name} digits after “0${radix}”`;
|
|
1481
|
+
}
|
|
1482
|
+
if (radix === 'x' && /[a-f]/v.test(digits) && !/[^\da-f_]/iv.test(digits)) {
|
|
1483
|
+
// The uppercase spelling is only suggested when it is valid, so a misplaced underscore is reported first, and a value outside the 64-bit range gets no example.
|
|
1484
|
+
if (!/^[\da-f]+(?:_[\da-f]+)*$/iv.test(digits)) {
|
|
1485
|
+
return 'An underscore in a number must be between two digits';
|
|
1486
|
+
}
|
|
1487
|
+
const uppercase = `0x${digits.toUpperCase()}`;
|
|
1488
|
+
return isValidValue(uppercase) ? `Hexadecimal digits are uppercase: ${abbreviate(uppercase)}` : 'Hexadecimal digits are uppercase';
|
|
1489
|
+
}
|
|
1490
|
+
// A lowercase hexadecimal digit is a digit in the wrong case, so the one named is a character that is no digit in either case.
|
|
1491
|
+
const validCharacter = { x: /[\da-f_]/iv, o: /[0-7_]/v, b: /[01_]/v }[radix];
|
|
1492
|
+
const character = [...digits].find(character => !validCharacter.test(character));
|
|
1493
|
+
return character === undefined ? 'An underscore in a number must be between two digits' : `Invalid ${name} digit “${character}”`;
|
|
1494
|
+
}
|
|
1495
|
+
if (/^-?\d[\d_]*E/v.test(text) || /^-?\d[\d_]*\.[\d_]+E/v.test(text)) {
|
|
1496
|
+
return 'An exponent marker is a lowercase “e”';
|
|
1497
|
+
}
|
|
1498
|
+
if (/^-?[\d_]+\.(?:$|[^\d_])/v.test(text)) {
|
|
1499
|
+
return 'A decimal point must be followed by a digit';
|
|
1500
|
+
}
|
|
1501
|
+
if (text.startsWith('-.')) {
|
|
1502
|
+
return 'A number cannot begin with “.”; write a digit before it, as in -0.5';
|
|
1503
|
+
}
|
|
1504
|
+
if (/^\d+(?::\d+)+$/v.test(text)) {
|
|
1505
|
+
return /^\d{1,2}:\d{2}(?::\d{2}(?:\.\d+)?)?$/v.test(text) ? 'A time of day is a string, so it must be quoted' : `Invalid number “${abbreviate(text)}”. A value that contains “:” must be quoted, as in '${abbreviate(text)}'`;
|
|
1506
|
+
}
|
|
1507
|
+
if (/^-?0[\d_]/v.test(text)) {
|
|
1508
|
+
// Removing the zero gives a valid number with a different meaning, so the message says what the zero usually meant.
|
|
1509
|
+
// A `'...'` string cannot hold a `'`, so then there is no example.
|
|
1510
|
+
const identifier = `an identifier, such as a ZIP code, as a string${unquotedText.includes('\'') ? '' : `: '${abbreviate(unquotedText)}'`}`;
|
|
1511
|
+
// The octal suggestion is only for the number on its own, not for one that more text follows, as in `0412 345 678`, and only when it is in range. It is decided on the whole number, because a cut can end before a digit that is not octal or that changes the value.
|
|
1512
|
+
const digits = fullText.replace(/^0+/v, '');
|
|
1513
|
+
const octal = `0o${digits === '' ? '0' : digits}`;
|
|
1514
|
+
if (unquotedText === fullText && /^0[0-7]+$/v.test(fullText) && isValidValue(octal)) {
|
|
1515
|
+
return `Leading zeros are not allowed in a decimal number. Write an octal number, such as a file mode, as ${abbreviate(octal)}, and ${identifier}`;
|
|
1516
|
+
}
|
|
1517
|
+
return /^0\d+$/v.test(text) ? `Leading zeros are not allowed in a decimal number. Write ${identifier}` : 'Leading zeros are not allowed in a decimal number';
|
|
1518
|
+
}
|
|
1519
|
+
if (/_(?:$|\D)|(?:^|\D)_/v.test(text)) {
|
|
1520
|
+
return 'An underscore in a number must be between two digits';
|
|
1521
|
+
}
|
|
1522
|
+
if (/^-?\d[\d_]*(?:\.[\d_]+)?e-?0_?\d/v.test(text)) {
|
|
1523
|
+
return 'Leading zeros are not allowed in an exponent';
|
|
1524
|
+
}
|
|
1525
|
+
if (/^-?\d[\d_]*(?:\.[\d_]+)?e-0$/v.test(text)) {
|
|
1526
|
+
return '“e-0” is not allowed, because an exponent of zero has one spelling: e0';
|
|
1527
|
+
}
|
|
1528
|
+
if (/e-?$/v.test(text)) {
|
|
1529
|
+
return 'Expected digits after the exponent marker “e”';
|
|
1530
|
+
}
|
|
1531
|
+
if (text.split('.').length > 2) {
|
|
1532
|
+
return `Invalid number “${abbreviate(text)}”. A value with several dots, such as a version number, must be quoted`;
|
|
1533
|
+
}
|
|
1534
|
+
if (/^-(?:\D|$)/v.test(text)) {
|
|
1535
|
+
if (/^-nan/iv.test(text)) {
|
|
1536
|
+
return 'NaN is not representable. Use null for a missing value';
|
|
1537
|
+
}
|
|
1538
|
+
return /^-inf/iv.test(text) ? `“${abbreviate(text)}” is not a value. Negative infinity is written -infinity` : 'Expected a digit or “infinity” after “-”';
|
|
1539
|
+
}
|
|
1540
|
+
return /[A-Za-z]/v.test(text) ? `Invalid number “${abbreviate(text)}”. A string value must be quoted${quotingExample(unquotedText)}` : `Invalid number “${abbreviate(text)}”`;
|
|
1541
|
+
}
|
|
1542
|
+
/*
|
|
1543
|
+
A date alone, which is not an instant. The instant it could be is only shown when the date exists, so that the example is valid.
|
|
1544
|
+
*/
|
|
1545
|
+
function describeDate(date) {
|
|
1546
|
+
const [year, month, day] = date.split('-').map(Number);
|
|
1547
|
+
const isExisting = year >= 1 && month >= 1 && month <= 12 && day >= 1 && day <= daysInMonth(year, month);
|
|
1548
|
+
return `${date} is a date, not an instant. Write a date as a string, as in '${date}'${isExisting ? `. An instant needs a time and an offset, as in ${date}T00:00:00Z` : ''}`;
|
|
1549
|
+
}
|
|
1550
|
+
/*
|
|
1551
|
+
`time` is the token after a space that follows `text`, or an empty string. `isWholeValue` is whether nothing that may be part of the instant follows `text`.
|
|
1552
|
+
*/
|
|
1553
|
+
function describeBadInstant(text, time, isWholeValue) {
|
|
1554
|
+
if (text.length > MAX_DIAGNOSED_LENGTH) {
|
|
1555
|
+
return `Invalid instant “${abbreviate(text)}”. ${INSTANT_FORMAT}`;
|
|
1556
|
+
}
|
|
1557
|
+
if (/^\d{4}-\d{2}-\d{2}$/v.test(text)) {
|
|
1558
|
+
return /^\d{2}:\d{2}/v.test(time) ? describeSpaceSeparatedInstant(text, time, isWholeValue) : describeDate(text);
|
|
1559
|
+
}
|
|
1560
|
+
if (/^\d{4}-\d{2}-\d{2}t/v.test(text)) {
|
|
1561
|
+
return 'The date and time separator in an instant is an uppercase “T”';
|
|
1562
|
+
}
|
|
1563
|
+
if (text.endsWith('z')) {
|
|
1564
|
+
return 'The UTC offset in an instant is an uppercase “Z”';
|
|
1565
|
+
}
|
|
1566
|
+
if (/[+\-]\d{4}$/v.test(text)) {
|
|
1567
|
+
return 'An instant\'s offset is written with a colon, as in +07:00';
|
|
1568
|
+
}
|
|
1569
|
+
if (!LOCAL_DATE_TIME.test(text)) {
|
|
1570
|
+
return `Invalid instant “${abbreviate(text)}”. ${INSTANT_FORMAT}`;
|
|
1571
|
+
}
|
|
1572
|
+
return isWholeValue ? `An instant needs an offset: Z or ±HH:MM. A date and time without an offset is a local time, which is not an instant, so write it as a string, as in '${abbreviate(text)}', or add the offset it was meant in` : 'An instant needs an offset: Z or ±HH:MM';
|
|
1573
|
+
}
|
|
1574
|
+
/*
|
|
1575
|
+
A date and a time with a space between them, where an instant has a "T". Following the example must not turn a local time into UTC silently, so a time without an offset is described as what it is. An instant is only shown when it is valid, so a date that does not exist, a time out of range, or an instant outside the years 0001 to 9999 in UTC gets the general format instead.
|
|
1576
|
+
*/
|
|
1577
|
+
function describeSpaceSeparatedInstant(date, time, isWholeValue) {
|
|
1578
|
+
const separator = 'The date and time separator in an instant is an uppercase “T”, not a space';
|
|
1579
|
+
const instant = `${date}T${time}`;
|
|
1580
|
+
if (instant.length > MAX_DIAGNOSED_LENGTH) {
|
|
1581
|
+
return `${separator}. ${INSTANT_FORMAT}`;
|
|
1582
|
+
}
|
|
1583
|
+
if (isValidValue(instant)) {
|
|
1584
|
+
return `${separator}, as in ${abbreviate(instant)}`;
|
|
1585
|
+
}
|
|
1586
|
+
return isWholeValue && LOCAL_DATE_TIME.test(instant) && isValidValue(`${instant}Z`) ? `${separator}, and an instant needs the offset it was meant in, as in ${abbreviate(instant)}Z for UTC. A date and time without an offset is a local time, which is not an instant, so write it as a string, as in '${abbreviate(`${date} ${time}`)}'` : `${separator}. ${INSTANT_FORMAT}`;
|
|
1587
|
+
}
|