soml-lang 0.0.2 → 0.0.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,43 +1,41 @@
1
1
  import { ParseError } from "./error.js";
2
- import { MAX_DEPTH, INT64_MIN, INT64_MAX, MIN_INSTANT, MAX_INSTANT, DURATION_UNITS, createDuration, requireTemporal, trimTrailingZeros, validateIntegersOption, describeCharacter, formatCodePoint, isBareKey, isBareKeyCharacter, findNumberEnd, findLineEnd, findBlockStringEnd, abbreviate, } from "./shared.js";
3
- const TAB = 0x09;
4
- const LF = 0x0A;
5
- const SPACE = 0x20;
6
- const DOUBLE_QUOTE = 0x22;
7
- const HASH = 0x23;
8
- const SINGLE_QUOTE = 0x27;
9
- const ASTERISK = 0x2A;
10
- const COMMA = 0x2C;
11
- const DASH = 0x2D;
12
- const DOT = 0x2E;
13
- const SLASH = 0x2F;
14
- const COLON = 0x3A;
15
- const OPEN_BRACKET = 0x5B;
16
- const CLOSE_BRACKET = 0x5D;
17
- const OPEN_BRACE = 0x7B;
18
- const CLOSE_BRACE = 0x7D;
19
- /*
20
- What may directly follow a scalar value: whitespace, a separator, a closing bracket, a comment, or the end.
21
- */
2
+ import { formatKey } from "./stringify.js";
3
+ import { MAX_DEPTH, INT64_MIN, INT64_MAX, MIN_INSTANT, MAX_INSTANT, DURATION_UNITS, createDuration, requireTemporal, trimTrailingZeros, getIntegersOption, describeCharacter, formatCodePoint, describeKey, isBareKeyCharacter, isSpace, skipSpaces, skipSpacesBack, isBlankLine, findNumberEnd, findLineEnd, findBlockStringEnd, abbreviate, LF, SPACE, DOUBLE_QUOTE, HASH, SINGLE_QUOTE, ASTERISK, COMMA, DASH, DOT, SLASH, COLON, OPEN_BRACKET, CLOSE_BRACKET, OPEN_BRACE, CLOSE_BRACE, } from "./shared.js";
22
4
  const valueEndCharacters = new Uint8Array(128);
23
5
  for (const character of ' \t\n,]}#/') {
24
6
  valueEndCharacters[character.codePointAt(0)] = 1;
25
7
  }
26
- const RADIX_DIGIT = {
27
- x: code => (code >= 0x30 && code <= 0x39) || (code >= 0x41 && code <= 0x46),
28
- o: code => code >= 0x30 && code <= 0x37,
29
- b: code => code === 0x30 || code === 0x31,
30
- };
31
8
  /*
32
- The longest token the diagnostic regular expressions look at. Their repeated groups need stack in proportion to the input, so a longer invalid token is described from its first part.
9
+ Whether a UTF-16 code unit may directly follow a scalar value: whitespace, a separator, a closing bracket, a comment, or the end, where `charCodeAt()` returns `NaN`.
10
+ */
11
+ function isValueEnd(code) {
12
+ return Number.isNaN(code) || (code < 128 && valueEndCharacters[code] === 1);
13
+ }
14
+ /*
15
+ Whether a UTF-16 code unit is a space, a tab, a line feed, or the end.
16
+ */
17
+ function isSpaceOrLineEnd(code) {
18
+ return isSpace(code) || code === LF || Number.isNaN(code);
19
+ }
20
+ /*
21
+ The lookup tables are maps rather than objects, so that a property added to `Object.prototype` is never read as an entry.
22
+ */
23
+ const RADIX_DIGIT = new Map([
24
+ ['x', code => (code >= 0x30 && code <= 0x39) || (code >= 0x41 && code <= 0x46)],
25
+ ['o', code => code >= 0x30 && code <= 0x37],
26
+ ['b', code => code === 0x30 || code === 0x31],
27
+ ]);
28
+ /*
29
+ The longest token that the instant regular expression and the diagnostic regular expressions look at. Their repeated groups need stack in proportion to the input, so a longer token is rejected with a general message, or described from its first part.
33
30
  */
34
31
  const MAX_DIAGNOSED_LENGTH = 1000;
35
32
  const RADIX_NAME = { x: 'hexadecimal', o: 'octal', b: 'binary' };
36
33
  const RADIX_ARTICLE = { x: 'A', o: 'An', b: 'A' };
37
34
  const INSTANT_PREFIX = /^\d{4}-\d{2}-\d{2}/v;
38
- const INSTANT_START = /^\d{4}-\d{2}-\d{2}T\d{2}:\d/v;
39
35
  const DURATION_UNIT_NAMES = DURATION_UNITS.keys().toArray();
40
36
  const INSTANT = /^(?<year>\d{4})-(?<month>\d{2})-(?<day>\d{2})T(?<hour>\d{2}):(?<minute>\d{2}):(?<second>\d{2})(?:\.(?<fraction>\d+))?(?<offset>Z|[+\-](?<offsetHour>\d{2}):(?<offsetMinute>\d{2}))$/v;
37
+ const LOCAL_DATE_TIME = /^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d+)?$/v;
38
+ const INSTANT_FORMAT = 'An instant is written as 2026-09-19T14:00:00Z, with an optional fraction of up to nine digits and an offset of Z or ±HH:MM';
41
39
  const UNICODE_ESCAPE = /u\{(?<hex>[\da-f]{1,6})\}/vy;
42
40
  /*
43
41
  Every C0 control character except tab and line feed, and DEL. They are errors anywhere in a document.
@@ -45,14 +43,23 @@ Every C0 control character except tab and line feed, and DEL. They are errors an
45
43
  // eslint-disable-next-line no-control-regex, regexp/no-control-character -- Control characters are exactly what must be found.
46
44
  const CONTROL_CHARACTER = /[\u{0}-\u{8}\u{B}-\u{1F}\u{7F}]/v;
47
45
  const LITERAL_STRING_END = /[\n']/gv;
48
- const BLANK_LINE = /^[\t ]*$/v;
46
+ const KEY_QUOTING_HINT = '. A key that contains characters other than letters, digits, “_”, and “-” must be quoted';
47
+ /*
48
+ The characters that may be meant as part of a key: visible ones that have no meaning in the grammar, which includes every bare key character. A `=` is left out, because after a key, as in `name=foo`, it was most likely meant as the `:` of INI and TOML, so a key that holds a `=` gets no quoting hint.
49
+ */
50
+ const QUOTABLE_KEY_CHARACTER = String.raw `[^\p{Default_Ignorable_Code_Point}\p{Other}\p{White_Space}"#'*,.\/:=\[\]\{\}]`;
51
+ const QUOTABLE_KEY_START = new RegExp(`^${QUOTABLE_KEY_CHARACTER}`, 'v');
52
+ /*
53
+ A key made of them, up to its `:`, so the quoting hint can show it quoted.
54
+ */
55
+ const QUOTABLE_KEY = new RegExp(`^${QUOTABLE_KEY_CHARACTER}+(?=:)`, 'v');
49
56
  const ESCAPED_STRING_SPECIAL = /[\n"\\]/gv;
50
- const SIMPLE_ESCAPES = {
51
- '\\': '\\',
52
- '"': '"',
53
- n: '\n',
54
- t: '\t',
55
- };
57
+ const SIMPLE_ESCAPES = new Map([
58
+ ['\\', '\\'],
59
+ ['"', '"'],
60
+ ['n', '\n'],
61
+ ['t', '\t'],
62
+ ]);
56
63
  /*
57
64
  What a backslash followed by one of these characters was probably meant to be.
58
65
  */
@@ -61,13 +68,34 @@ const ESCAPE_MISTAKES = new Map([
61
68
  ['\'', 'A \' needs no escape inside "..."'],
62
69
  ['\n', String.raw `A backslash must be followed by an escape character. Use \\ for a literal backslash, or a '...' string`],
63
70
  ]);
64
- export function parse(text, options = {}) {
65
- const { integers = 'bigint' } = options;
66
- validateIntegersOption(integers);
71
+ export function parse(text, options) {
72
+ const integers = getIntegersOption(options);
67
73
  const source = decode(text);
68
74
  checkCharacters(source);
75
+ // The default `createTime` makes a `Temporal` object of every instant and duration, so no `Time` is left.
69
76
  return new Parser(source, integers).parseDocument();
70
77
  }
78
+ /*
79
+ An instant, as nanoseconds since the Unix epoch, or a duration, as its length in nanoseconds.
80
+ */
81
+ export class Time {
82
+ type;
83
+ nanoseconds;
84
+ constructor(type, nanoseconds) {
85
+ this.type = type;
86
+ this.nanoseconds = nanoseconds;
87
+ }
88
+ toTemporal() {
89
+ return this.type === 'Instant' ? new (requireTemporal().Instant)(this.nanoseconds) : createDuration(this.nanoseconds);
90
+ }
91
+ }
92
+ /*
93
+ The same as `parse()` for a string, except that each instant and duration is a `Time` instead of a `Temporal` object, so that it works without `Temporal`. For the tree, which makes the `Temporal` object only when the value is read.
94
+ */
95
+ export function parseWithTimes(text) {
96
+ checkCharacters(text);
97
+ return new Parser(text, 'bigint', time => time).parseDocument();
98
+ }
71
99
  const typedArrayTag = Object.getOwnPropertyDescriptor(Object.getPrototypeOf(Uint8Array.prototype), Symbol.toStringTag).get;
72
100
  function decode(text) {
73
101
  if (typeof text === 'string') {
@@ -170,27 +198,19 @@ function defineMember(object, key, value) {
170
198
  object[key] = value;
171
199
  }
172
200
  }
173
- function describeValue(value) {
174
- if (Array.isArray(value)) {
175
- return 'an array';
176
- }
177
- // Every parsed object is created with `{}`, unlike an instant or a duration.
178
- return value !== null && typeof value === 'object' && Object.getPrototypeOf(value) === Object.prototype ? 'an object' : 'a value';
179
- }
180
- function formatPath(path) {
181
- return path.map(segment => isBareKey(segment) ? abbreviate(segment) : JSON.stringify(abbreviate(segment))).join('.');
182
- }
183
201
  class Parser {
184
202
  #source;
185
203
  #index = 0;
186
204
  #integers;
205
+ #createTime;
187
206
  /*
188
- Objects created by dotted keys. They may be extended by further dotted keys, while an object written with braces is closed.
207
+ Where the document's collection starts, after the comments and whitespace before it.
189
208
  */
190
- #dottedObjects = new WeakSet();
191
- constructor(source, integers) {
209
+ #documentStart = 0;
210
+ constructor(source, integers, createTime = time => time.toTemporal()) {
192
211
  this.#source = source;
193
212
  this.#integers = integers;
213
+ this.#createTime = createTime;
194
214
  }
195
215
  #fail(reason, offset = this.#index) {
196
216
  throw ParseError.create(reason, this.#source, offset);
@@ -218,11 +238,11 @@ class Parser {
218
238
  }
219
239
  let hasLineBreak = this.#skipTrivia();
220
240
  while (!this.#isAtEnd()) {
241
+ if (this.#code() === COMMA) {
242
+ this.#fail('Top-level entries are separated by line breaks, not commas. Use braces for a one-line object');
243
+ }
221
244
  if (!hasLineBreak) {
222
- if (this.#code() === COMMA) {
223
- this.#fail('Top-level entries are separated by line breaks, not commas. Use braces for a one-line object');
224
- }
225
- this.#fail(`Expected a line break before the next entry, but found ${this.#describeHere()}`);
245
+ this.#fail(`Expected a line break before the next entry, but found ${this.#describeHere()}${this.#slashCommentHint()}`);
226
246
  }
227
247
  this.#parseEntry(object, 1);
228
248
  hasLineBreak = this.#skipTrivia();
@@ -230,21 +250,18 @@ class Parser {
230
250
  return object;
231
251
  }
232
252
  /*
233
- A first line that is one scalar, such as `5` or an instant, fails as an entry. That failure is reported as what it is.
253
+ A whole document that is one scalar, such as `5` or an instant, fails as an entry. That failure is reported as what it is.
234
254
  */
235
255
  #diagnoseBareValue(start) {
236
256
  this.#index = start;
237
257
  let isBareValue = false;
238
258
  try {
239
259
  this.#parseValue(1);
240
- isBareValue = this.#skipTrivia() || this.#isAtEnd();
260
+ this.#skipTrivia();
261
+ isBareValue = this.#isAtEnd();
241
262
  }
242
- catch (error) {
243
- // A bare key cannot hold a ":", so a token shaped like an instant was meant as one, and its own error says what is wrong with it.
244
- if (INSTANT_START.test(this.#source.slice(start, start + 15))) {
245
- throw error;
246
- }
247
- // Otherwise it is not a bare value either, and the original error stands.
263
+ catch {
264
+ // Not a bare value either, so the original error stands.
248
265
  }
249
266
  if (isBareValue) {
250
267
  this.#fail('A bare value is not a document. A document is an object or an array, so write it as `key: value` or `[value]`', start);
@@ -258,7 +275,7 @@ class Parser {
258
275
  let hasCrossedLineBreak = false;
259
276
  for (;;) {
260
277
  const code = source.charCodeAt(this.#index);
261
- if (code === SPACE || code === TAB) {
278
+ if (isSpace(code)) {
262
279
  this.#index++;
263
280
  }
264
281
  else if (code === LF) {
@@ -296,7 +313,7 @@ class Parser {
296
313
  const nested = source.indexOf('/*', start + 2);
297
314
  // The body may not contain `/*`. An opening that overlaps the closing `*/`, as in `/*/`, is not inside the body.
298
315
  if (nested !== -1 && nested + 2 <= end) {
299
- this.#fail('Block comments cannot be nested, and their body may not contain "/*"', nested);
316
+ this.#fail('Block comments cannot be nested, and their body may not contain “/*”', nested);
300
317
  }
301
318
  this.#index = end + 2;
302
319
  }
@@ -305,59 +322,155 @@ class Parser {
305
322
  */
306
323
  #parseEntry(object, depth) {
307
324
  const keyStart = this.#index;
308
- const path = this.#parseKey();
309
- if (depth + path.length - 1 > MAX_DEPTH) {
310
- this.#fail(`The document is nested more than ${MAX_DEPTH} levels deep`, keyStart);
311
- }
325
+ const key = this.#parseKey();
312
326
  const code = this.#code();
327
+ if (code === DOT) {
328
+ this.#failDotInKey(keyStart);
329
+ }
313
330
  if (code !== COLON) {
314
331
  this.#failMissingColon(keyStart);
315
332
  }
333
+ const colon = this.#index;
316
334
  this.#index++;
317
335
  this.#skipTrivia();
318
- const value = this.#parseValue(depth + path.length);
319
- this.#assign(object, path, value, keyStart);
336
+ let value;
337
+ try {
338
+ value = this.#parseValue(depth + 1);
339
+ }
340
+ catch (error) {
341
+ if (error instanceof ParseError) {
342
+ this.#diagnoseBadValue(error, key, keyStart, colon);
343
+ }
344
+ throw error;
345
+ }
346
+ if (Object.hasOwn(object, key)) {
347
+ this.#fail(`Duplicate key ${describeKey(key)}`, keyStart);
348
+ }
349
+ defineMember(object, key, value);
320
350
  }
321
- #failMissingColon(keyStart) {
351
+ /*
352
+ Reports a value that failed to parse as what the entry was meant to be, when that is clear.
353
+ */
354
+ #diagnoseBadValue(error, key, keyStart, colon) {
355
+ // A key that contains a `:`, as in `12:30: 'lunch'`, ends at the first `:`, and the rest is read as the value.
356
+ if (isBareKeyCharacter(this.#code(colon + 1))) {
357
+ this.#diagnoseKeyWithColon(keyStart);
358
+ }
359
+ this.#index = colon + 1;
360
+ const isValueOnNextLine = this.#skipTrivia();
361
+ const valueStart = this.#index;
362
+ const commentHint = this.#describeCommentAsValue(colon);
363
+ if (isValueOnNextLine) {
364
+ this.#diagnoseMissingValue(valueStart, key, keyStart, commentHint);
365
+ }
366
+ // A value that was left out at the end of the document or of an object.
367
+ const code = this.#code(valueStart);
368
+ if (commentHint !== '' && error.offset === valueStart && (code === CLOSE_BRACE || Number.isNaN(code))) {
369
+ this.#fail(`${error.reason}${commentHint}`, valueStart);
370
+ }
371
+ }
372
+ /*
373
+ A bare key followed directly by `:` and more of a key, up to a `:` that ends the key, as in `a:b: 1`.
374
+ */
375
+ #diagnoseKeyWithColon(keyStart) {
322
376
  const source = this.#source;
323
- let next = this.#index;
324
- while (source.charCodeAt(next) === SPACE || source.charCodeAt(next) === TAB) {
325
- next++;
377
+ for (let index = keyStart; index < keyStart + MAX_DIAGNOSED_LENGTH; index++) {
378
+ const code = source.charCodeAt(index);
379
+ if (code === COLON) {
380
+ const next = source.charCodeAt(index + 1);
381
+ if (next === LF || isSpace(next) || Number.isNaN(next)) {
382
+ this.#fail(`A key that contains “:” must be quoted, as in '${abbreviate(source.slice(keyStart, index))}'`, keyStart);
383
+ }
384
+ }
385
+ else if (!isBareKeyCharacter(code)) {
386
+ return;
387
+ }
388
+ }
389
+ }
390
+ /*
391
+ The hint for a `#` directly after a `:`, as in `color: #FFF`, which starts a comment rather than a value.
392
+ */
393
+ #describeCommentAsValue(colon) {
394
+ const source = this.#source;
395
+ const hash = skipSpaces(source, colon + 1);
396
+ if (source.charCodeAt(hash) !== HASH) {
397
+ return '';
326
398
  }
399
+ // A `#` followed by a space, or by another `#`, starts an ordinary comment. A separator or a closing bracket after the value is not part of it. The regular expression, which needs stack in proportion to its match, only sees the start of a long comment, which is cut short in the message anyway.
400
+ const match = /^#[^\t\n #,\]\}][^\t\n ,\]\}]*/v.exec(source.slice(hash, hash + MAX_DIAGNOSED_LENGTH));
401
+ return match === null ? '' : `. “#” starts a comment, so a value that starts with “#” must be quoted${quotingExample(match[0])}`;
402
+ }
403
+ /*
404
+ An entry whose value was left out, as in `a:` followed by `'b': 1` on the next line, reads the next key as the value. That failure is reported as what it is.
405
+ */
406
+ #diagnoseMissingValue(start, parent, keyStart, commentHint) {
407
+ // A YAML block sequence.
408
+ if (this.#code(start) === DASH && isSpace(this.#code(start + 1))) {
409
+ this.#fail('Expected a value, but found a “-” list. An array is written in brackets, as in [80, 443]', start);
410
+ }
411
+ this.#index = start;
412
+ let key;
413
+ try {
414
+ key = this.#parseKey();
415
+ }
416
+ catch {
417
+ // Not a key either, so the original error stands.
418
+ }
419
+ if (key === undefined || this.#code() !== COLON) {
420
+ return;
421
+ }
422
+ // Whitespace or the end follows the `:` of a bare key. A digit follows the `:` in an instant such as `2026-09-19T25:00:00Z`, whose own error is more precise. A quoted key cannot be part of a value, so anything may follow its `:`.
423
+ const next = this.#code(this.#index + 1);
424
+ const firstCode = this.#code(start);
425
+ if (firstCode !== SINGLE_QUOTE && firstCode !== DOUBLE_QUOTE && next !== LF && !isSpace(next) && !Number.isNaN(next)) {
426
+ return;
427
+ }
428
+ const hint = commentHint === '' ? this.#describeIndentedKey(parent, key, keyStart, start) : commentHint;
429
+ this.#fail(`Expected a value, but found the key ${describeKey(key)}${hint}`, start);
430
+ }
431
+ /*
432
+ The hint for a key at `start` that is indented under the entry whose value is missing, as YAML nests an object.
433
+ */
434
+ #describeIndentedKey(parent, key, keyStart, start) {
435
+ const keyIndentation = getIndentation(this.#source, keyStart);
436
+ const indentation = getIndentation(this.#source, start);
437
+ // The keys are written as in a document, so that the suggestion is valid, rather than as JSON strings like the rest of the message, whose escapes, such as `\b`, are not all valid.
438
+ const hint = `. Indentation does not nest objects, so write ${abbreviate(formatKey(parent), 200)}: {${abbreviate(formatKey(key), 200)}: …}`;
439
+ return keyIndentation === undefined || indentation === undefined || indentation <= keyIndentation ? '' : hint;
440
+ }
441
+ #failMissingColon(keyStart) {
442
+ const source = this.#source;
443
+ const next = skipSpaces(source, this.#index);
327
444
  const nextCode = source.charCodeAt(next);
328
445
  if (nextCode === COLON && next > this.#index) {
329
- this.#fail('Whitespace is not allowed between a key and its ":"');
446
+ this.#fail('Whitespace is not allowed between a key and its “:”');
330
447
  }
331
- // Nothing else can follow a key, so a block comment there was meant to come before the ":".
448
+ // A block comment followed by the ":" was meant to come before it.
332
449
  if (nextCode === SLASH && source.charCodeAt(next + 1) === ASTERISK) {
333
- this.#fail('A comment is not allowed between a key and its ":"', next);
450
+ const commentEnd = source.indexOf('*/', next + 2);
451
+ if (commentEnd !== -1 && source.charCodeAt(skipSpaces(source, commentEnd + 2)) === COLON) {
452
+ this.#fail('A comment is not allowed between a key and its “:”', next);
453
+ }
334
454
  }
335
455
  const colon = source.indexOf(':', next);
336
- // A key with a space in it, such as `the name: 1`. The hint is only given for plain words, because quoting a dotted or quoted key would change what it means.
337
- if (colon !== -1 && next > this.#index && isBareKeyCharacter(nextCode) && colon < findLineEnd(source, next)) {
338
- const key = source.slice(keyStart, colon).trimEnd();
456
+ // A key with a space in it, such as `the name: 1`. The hint is only given for plain words, because quoting a quoted key would change what it means, and for a `:` that whitespace or the end follows, because quoting the words before the `:` in `server localhost:8080` would give a valid document with another meaning. So `the name:1` gets no hint.
457
+ if (colon !== -1 && next > this.#index && isBareKeyCharacter(nextCode) && colon < findLineEnd(source, next) && isSpaceOrLineEnd(source.charCodeAt(colon + 1))) {
458
+ // Only spaces and tabs are trimmed, because `trimEnd()` would also remove characters that are not whitespace in SOML, such as U+00A0.
459
+ const key = source.slice(keyStart, skipSpacesBack(source, colon));
339
460
  if (isWordsWithSpaces(key)) {
340
461
  this.#fail(`A bare key cannot contain spaces. Quote it, as in '${abbreviate(key)}'`, keyStart);
341
462
  }
342
463
  }
343
464
  if (this.#isAtEnd() || this.#code() === LF) {
344
- this.#fail('Expected ":" after the key');
465
+ this.#fail('Expected “:” after the key');
345
466
  }
346
467
  // A character directly after a bare key is most likely meant to be part of it.
347
468
  const previousCode = source.charCodeAt(this.#index - 1);
348
- const hint = previousCode !== SINGLE_QUOTE && previousCode !== DOUBLE_QUOTE && next === this.#index ? '. A key that contains characters other than letters, digits, "_", and "-" must be quoted' : '';
469
+ const hint = previousCode !== SINGLE_QUOTE && previousCode !== DOUBLE_QUOTE && next === this.#index && QUOTABLE_KEY_START.test(source.slice(next, next + 2)) ? KEY_QUOTING_HINT : '';
349
470
  this.#index = next;
350
- this.#fail(`Expected ":" after the key, but found ${this.#describeHere()}${hint}`);
471
+ this.#fail(`Expected “:” after the key, but found ${this.#describeHere()}${hint}`);
351
472
  }
352
473
  #parseKey() {
353
- const path = [this.#parseKeySegment()];
354
- while (this.#code() === DOT) {
355
- this.#index++;
356
- path.push(this.#parseKeySegment(true));
357
- }
358
- return path;
359
- }
360
- #parseKeySegment(isAfterDot = false) {
361
474
  const source = this.#source;
362
475
  const start = this.#index;
363
476
  const code = source.charCodeAt(start);
@@ -373,100 +486,93 @@ class Parser {
373
486
  end++;
374
487
  }
375
488
  if (end === start) {
376
- if (isAfterDot) {
377
- this.#fail(this.#isAtEnd() ? 'Expected a key segment after "."' : `Expected a key segment after ".", but found ${this.#describeHere()}`);
489
+ if (code === OPEN_BRACKET) {
490
+ this.#diagnoseTableHeader();
378
491
  }
379
- this.#fail(this.#isAtEnd() ? 'Expected a key' : `Expected a key, but found ${this.#describeHere()}`);
492
+ this.#fail(this.#isAtEnd() ? 'Expected a key' : `Expected a key, but found ${this.#describeHere()}${this.#keyQuotingHint()}${this.#slashCommentHint()}`);
380
493
  }
381
494
  this.#index = end;
382
495
  return source.slice(start, end);
383
496
  }
384
- #assign(object, path, value, keyStart) {
385
- let target = object;
386
- const lastIndex = path.length - 1;
387
- for (let index = 0; index < lastIndex; index++) {
388
- const key = path[index];
389
- if (!Object.hasOwn(target, key)) {
390
- const child = {};
391
- this.#dottedObjects.add(child);
392
- defineMember(target, key, child);
393
- target = child;
394
- continue;
395
- }
396
- // Only an object built by dotted keys is in the set, which is checked next.
397
- const existing = target[key];
398
- if (!this.#dottedObjects.has(existing)) {
399
- const prefix = formatPath(path.slice(0, index + 1));
400
- const what = describeValue(existing);
401
- const reason = what === 'an object' ? 'an object written with braces, which is closed' : what;
402
- this.#fail(`Cannot set ${formatPath(path)}, because ${prefix} is already ${reason} and a dotted key cannot extend it`, keyStart);
403
- }
404
- target = existing;
497
+ /*
498
+ The hint for a key that starts with a character a bare key cannot hold, such as the `$` in `$schema`, or nothing for a character with another meaning, such as `}`.
499
+ */
500
+ #keyQuotingHint() {
501
+ if (!QUOTABLE_KEY_START.test(this.#source.slice(this.#index, this.#index + 2))) {
502
+ return '';
405
503
  }
406
- const key = path[lastIndex];
407
- if (Object.hasOwn(target, key)) {
408
- if (this.#dottedObjects.has(target[key])) {
409
- this.#fail(`Cannot set ${formatPath(path)}: it is already an object built by dotted keys`, keyStart);
410
- }
411
- this.#fail(`Duplicate key ${formatPath(path)}`, keyStart);
504
+ // A longer key gets the hint without the example, and the regular expression, which backtracks, never sees a huge run.
505
+ const key = QUOTABLE_KEY.exec(this.#source.slice(this.#index, this.#index + MAX_DIAGNOSED_LENGTH))?.[0];
506
+ return key === undefined ? KEY_QUOTING_HINT : `${KEY_QUOTING_HINT}, as in '${abbreviate(key)}'`;
507
+ }
508
+ /*
509
+ A `.` after a key, as in `example.com: 1`. A key is never a path, so the `.` is meant either as part of the key or as nesting.
510
+ */
511
+ #failDotInKey(keyStart) {
512
+ const source = this.#source;
513
+ let end = this.#index;
514
+ while (end < keyStart + MAX_DIAGNOSED_LENGTH && (isBareKeyCharacter(source.charCodeAt(end)) || source.charCodeAt(end) === DOT)) {
515
+ end++;
516
+ }
517
+ const key = source.slice(keyStart, end);
518
+ const words = key.split('.');
519
+ // The suggestions are only given for a bare key whose dots each have a word on both sides, so that both are valid.
520
+ if (source.charCodeAt(end) !== COLON || !isBareKeyCharacter(source.charCodeAt(keyStart)) || words.includes('')) {
521
+ this.#fail('A key cannot contain “.” unless it is quoted. Quote the whole key, or use braces to nest, as in a: {b: …}');
412
522
  }
413
- defineMember(target, key, value);
523
+ const nested = `${words.join(': {')}: …${'}'.repeat(words.length - 1)}`;
524
+ this.#fail(`A bare key cannot contain “.”. Quote it, as in '${abbreviate(key)}', or use braces to nest, as in ${abbreviate(nested)}`);
414
525
  }
415
526
  #parseObject(depth) {
416
- if (depth > MAX_DEPTH) {
417
- this.#fail(`The document is nested more than ${MAX_DEPTH} levels deep`);
418
- }
419
- const start = this.#index;
420
527
  const object = {};
421
- this.#index++;
422
- this.#skipTrivia();
423
- for (;;) {
424
- const code = this.#code();
425
- if (code === CLOSE_BRACE) {
426
- this.#index++;
427
- return object;
428
- }
429
- if (this.#isAtEnd()) {
430
- this.#fail('Unterminated object: expected "}"', start);
431
- }
528
+ this.#parseItems(depth, CLOSE_BRACE, () => {
432
529
  this.#parseEntry(object, depth);
433
- this.#skipTrivia();
434
- if (this.#code() === COMMA) {
435
- this.#index++;
436
- this.#skipTrivia();
437
- }
438
- else if (this.#code() !== CLOSE_BRACE) {
439
- this.#fail(this.#isAtEnd() ? 'Unterminated object: expected "}"' : `Expected "," or "}" after an object member, but found ${this.#describeHere()}`, this.#isAtEnd() ? start : this.#index);
440
- }
441
- }
530
+ });
531
+ return object;
442
532
  }
443
533
  #parseArray(depth) {
534
+ const array = [];
535
+ this.#parseItems(depth, CLOSE_BRACKET, () => {
536
+ array.push(this.#parseValue(depth + 1));
537
+ });
538
+ return array;
539
+ }
540
+ /*
541
+ `{` or `[` at depth `depth`, then items separated by a comma, a line break, or both, with an optional trailing comma, then the `closing` bracket. A comma must be on the line of the item before it.
542
+ */
543
+ #parseItems(depth, closing, parseItem) {
444
544
  if (depth > MAX_DEPTH) {
445
545
  this.#fail(`The document is nested more than ${MAX_DEPTH} levels deep`);
446
546
  }
447
547
  const start = this.#index;
448
- const array = [];
449
548
  this.#index++;
450
549
  this.#skipTrivia();
451
- for (;;) {
452
- const code = this.#code();
453
- if (code === CLOSE_BRACKET) {
454
- this.#index++;
455
- return array;
550
+ while (this.#code() !== closing) {
551
+ let hasLineBreak = false;
552
+ let itemEnd = this.#index;
553
+ if (!this.#isAtEnd()) {
554
+ parseItem();
555
+ itemEnd = this.#index;
556
+ hasLineBreak = this.#skipTrivia();
456
557
  }
457
558
  if (this.#isAtEnd()) {
458
- this.#fail('Unterminated array: expected "]"', start);
559
+ this.#fail(`Unterminated ${closing === CLOSE_BRACE ? 'object' : 'array'}: expected “${String.fromCharCode(closing)}”`, start);
459
560
  }
460
- array.push(this.#parseValue(depth + 1));
461
- this.#skipTrivia();
462
561
  if (this.#code() === COMMA) {
562
+ if (hasLineBreak) {
563
+ this.#fail('A comma must be on the same line as the item before it. The line break already separates the items, so remove the comma');
564
+ }
463
565
  this.#index++;
464
566
  this.#skipTrivia();
465
567
  }
466
- else if (this.#code() !== CLOSE_BRACKET) {
467
- this.#fail(this.#isAtEnd() ? 'Unterminated array: expected "]"' : `Expected "," or "]" after an array item, but found ${this.#describeHere()}`, this.#isAtEnd() ? start : this.#index);
568
+ else if (!hasLineBreak && this.#code() !== closing) {
569
+ const item = closing === CLOSE_BRACE ? 'an object member' : 'an array item';
570
+ // Without a line break that separates, a line break in the gap is inside a block comment.
571
+ const hint = this.#source.slice(itemEnd, this.#index).includes('\n') ? '. A line break inside a block comment does not separate items' : this.#slashCommentHint();
572
+ this.#fail(`Expected “,”, a line break, or “${String.fromCharCode(closing)}” after ${item}, but found ${this.#describeHere()}${hint}`);
468
573
  }
469
574
  }
575
+ this.#index++;
470
576
  }
471
577
  #parseValue(depth) {
472
578
  const code = this.#code();
@@ -476,7 +582,10 @@ class Parser {
476
582
  if (code === OPEN_BRACKET) {
477
583
  return this.#parseArray(depth);
478
584
  }
585
+ const start = this.#index;
479
586
  let value;
587
+ // An int or a float written with digits.
588
+ let isNumber = false;
480
589
  if (code === SINGLE_QUOTE || code === DOUBLE_QUOTE) {
481
590
  value = this.#parseString();
482
591
  }
@@ -497,16 +606,55 @@ class Parser {
497
606
  }
498
607
  else if (code === DASH || isDigit(code)) {
499
608
  value = this.#parseNumberOrInstant();
609
+ isNumber = typeof value === 'bigint' || typeof value === 'number';
500
610
  }
501
611
  else {
502
612
  this.#failUnexpectedValue();
503
613
  }
504
- const next = this.#code();
505
- if (!this.#isAtEnd() && (next >= 128 || valueEndCharacters[next] !== 1)) {
506
- this.#fail(`Unexpected ${this.#describeHere()} after a value`);
614
+ if (!isValueEnd(this.#code())) {
615
+ this.#failAfterValue(start, isNumber);
616
+ }
617
+ if (isNumber && isSpace(this.#code())) {
618
+ this.#diagnoseUnitAfterSpace(start);
507
619
  }
508
620
  return value;
509
621
  }
622
+ /*
623
+ A character that cannot follow the value that starts at `start`.
624
+ */
625
+ #failAfterValue(start, isNumber) {
626
+ const code = this.#code();
627
+ // Go writes microseconds as `µs`, with the micro sign or the Greek letter mu. The duration is only suggested when it is valid, which a number with an exponent or a radix prefix, for example, is not.
628
+ if (isNumber && (code === 0xB5 || code === 0x3_BC) && this.#code(this.#index + 1) === 0x73 /* s */) {
629
+ const number = this.#source.slice(start, this.#index);
630
+ this.#fail(`The unit for microseconds is written us${isValidValue(`${number}us`) ? `, as in ${abbreviate(number)}us` : ''}`);
631
+ }
632
+ // A `''` inside a '...' string, as SQL and YAML escape a quote, ends the string.
633
+ if (code === SINGLE_QUOTE && this.#code(start) === SINGLE_QUOTE) {
634
+ this.#fail('There is no \'\' escape in a \'...\' string. Write a string that contains \' as "...", as in "it\'s"');
635
+ }
636
+ this.#fail(`Unexpected ${this.#describeHere()} after a value`);
637
+ }
638
+ /*
639
+ A number followed by a space and a word, such as `512 MiB` or `10 seconds`, which is an error anyway. A word followed by more than spaces, a separator, or a comment is left to the general errors, because it may be a key, as in `a: 1 b: 2`, or a sentence.
640
+ */
641
+ #diagnoseUnitAfterSpace(start) {
642
+ const source = this.#source;
643
+ const unitStart = skipSpaces(source, this.#index);
644
+ let unitEnd = unitStart;
645
+ while (isLetter(source.charCodeAt(unitEnd))) {
646
+ unitEnd++;
647
+ }
648
+ const unit = source.slice(unitStart, unitEnd);
649
+ // A keyword after a number is a missing comma.
650
+ if (unit === '' || !isValueEnd(source.charCodeAt(skipSpaces(source, unitEnd))) || ['true', 'false', 'null', 'infinity'].includes(unit)) {
651
+ return;
652
+ }
653
+ const number = source.slice(start, this.#index);
654
+ // The duration is only suggested when it is valid, which a number with an exponent or a radix prefix, a fraction of a nanosecond, a negative zero, or a value outside the 64-bit range is not. The string is always offered, because a unit such as `m` may mean meters rather than minutes.
655
+ const duration = DURATION_UNITS.has(unit) && isValidValue(`${number}${unit}`) ? `${abbreviate(number)}${unit}` : '10s';
656
+ this.#fail(`A unit cannot follow a number after a space. Write a duration without the space, as in ${duration}, and anything else, such as a size, as a string, as in '${abbreviate(`${number} ${unit}`)}'`, unitStart);
657
+ }
510
658
  #isKeyword(word) {
511
659
  if (!this.#source.startsWith(word, this.#index)) {
512
660
  return false;
@@ -525,10 +673,11 @@ class Parser {
525
673
  }
526
674
  const code = this.#code();
527
675
  if (code === 0x2B /* + */) {
528
- this.#fail('A "+" sign is not allowed. A number without a sign is positive');
676
+ this.#fail('A “+” sign is not allowed. A number without a sign is positive');
529
677
  }
678
+ // A `.` that no digit follows begins a string, such as `.env` or `./foo`, rather than a number.
530
679
  if (code === DOT) {
531
- this.#fail('A number cannot begin with "."; write a digit before it, as in 0.5');
680
+ this.#fail(isDigit(this.#code(this.#index + 1)) ? 'A number cannot begin with “.”; write a digit before it, as in 0.5' : describeUnknownWord('.', this.#unquotedText()));
532
681
  }
533
682
  let wordEnd = this.#index;
534
683
  while (isBareKeyCharacter(this.#code(wordEnd))) {
@@ -536,13 +685,66 @@ class Parser {
536
685
  }
537
686
  if (wordEnd > this.#index) {
538
687
  const word = this.#source.slice(this.#index, wordEnd);
539
- // An entry whose value was left out, as in `a:` followed by `b: 1` on the next line, reads the next key as the value.
540
- if (this.#code(wordEnd) === COLON) {
688
+ // A key where a value should be, as when the value of an entry is left out and the next line has a key, or as in `[a: 1]`, is reported as a key. A word after the `:` of a member and a space, as in `msg: Error: file not found` or `url: https://example.com`, is that member's value instead, an unquoted string. Without the space, as in `a:b: 1`, the first `:` was most likely meant as part of the key.
689
+ const spacesStart = skipSpacesBack(this.#source, this.#index);
690
+ const isMemberValue = spacesStart < this.#index && this.#code(spacesStart - 1) === COLON;
691
+ if (!isMemberValue && this.#code(wordEnd) === COLON) {
541
692
  this.#fail(`Expected a value, but found the key ${abbreviate(word)}`);
542
693
  }
543
- this.#fail(describeUnknownWord(word));
694
+ this.#diagnoseTableHeader(true);
695
+ this.#fail(describeUnknownWord(word, this.#unquotedText()));
696
+ }
697
+ this.#fail(`Expected a value, but found ${this.#describeHere()}${this.#isBlockScalarIndicator() ? '. Write a multiline string as a block string, between \'\'\' lines' : this.#slashCommentHint()}`);
698
+ }
699
+ /*
700
+ The text from `start` that was most likely meant as one unquoted string, such as `John Smith`: up to the end of the line, a comma, a closing bracket, or a comment after a space or a tab, without the spaces and tabs at its end.
701
+ */
702
+ #unquotedText(start = this.#index) {
703
+ const source = this.#source;
704
+ const limit = Math.min(source.length, start + MAX_DIAGNOSED_LENGTH);
705
+ let end = start;
706
+ while (end < limit && !isUnquotedTextEnd(source, end)) {
707
+ end++;
544
708
  }
545
- this.#fail(`Expected a value, but found ${this.#describeHere()}`);
709
+ return source.slice(start, skipSpacesBack(source, end));
710
+ }
711
+ /*
712
+ A TOML table header, as in `[server]` or `[[servers]]` on a line of its own, reads as an array that holds a word, or as a key that starts with "[". Where a value is expected, `isValue`, it is only a table header when its bracket opens the document, because a table cannot be written as an item of an array inside it.
713
+ */
714
+ #diagnoseTableHeader(isValue = false) {
715
+ const source = this.#source;
716
+ const lineStart = source.lastIndexOf('\n', this.#index - 1) + 1;
717
+ if (isValue && skipSpaces(source, lineStart) !== this.#documentStart) {
718
+ return;
719
+ }
720
+ const line = source.slice(lineStart, Math.min(findLineEnd(source, this.#index), lineStart + MAX_DIAGNOSED_LENGTH));
721
+ // The name is a valid key, so that the suggestion is valid.
722
+ const match = /^[\t ]*\[\[?(?<name>[A-Z_a-z][\w\-]*)\]\]?[\t ]*$/v.exec(line);
723
+ if (match === null) {
724
+ return;
725
+ }
726
+ const { name } = match.groups;
727
+ this.#fail(`There are no table headers. Write the table as an object, as in ${abbreviate(name)}: {…}`);
728
+ }
729
+ /*
730
+ A YAML literal block scalar indicator, as in `key: |` or `key: |-`, at the end of its line.
731
+ */
732
+ #isBlockScalarIndicator() {
733
+ const code = this.#code();
734
+ // A folded block scalar, `>`, joins its lines, which a block string does not, so only `|` gets the hint.
735
+ if (code !== 0x7C /* | */) {
736
+ return false;
737
+ }
738
+ const chomping = this.#code(this.#index + 1);
739
+ const end = skipSpaces(this.#source, chomping === DASH || chomping === 0x2B /* + */ ? this.#index + 2 : this.#index + 1);
740
+ const next = this.#code(end);
741
+ return next === LF || Number.isNaN(next);
742
+ }
743
+ /*
744
+ The hint for a `//` comment, as in JavaScript.
745
+ */
746
+ #slashCommentHint() {
747
+ return this.#code() === SLASH && this.#code(this.#index + 1) === SLASH ? '. A comment starts with “#”' : '';
546
748
  }
547
749
  #describeHere() {
548
750
  if (this.#isAtEnd()) {
@@ -602,8 +804,9 @@ class Parser {
602
804
  #parseEscape(offset) {
603
805
  const source = this.#source;
604
806
  const character = source[offset + 1];
605
- if (character !== undefined && Object.hasOwn(SIMPLE_ESCAPES, character)) {
606
- return { text: SIMPLE_ESCAPES[character], end: offset + 2 };
807
+ const simple = SIMPLE_ESCAPES.get(character ?? '');
808
+ if (simple !== undefined) {
809
+ return { text: simple, end: offset + 2 };
607
810
  }
608
811
  if (character === 'u') {
609
812
  UNICODE_ESCAPE.lastIndex = offset + 1;
@@ -612,12 +815,10 @@ class Parser {
612
815
  this.#fail(describeBadUnicodeEscape(source, offset + 1), offset);
613
816
  }
614
817
  const { hex } = match.groups;
615
- if (hex.length > 1 && hex.startsWith('0')) {
616
- this.#fail(String.raw `A Unicode escape may not have leading zeros; write \u{${hex.replace(/^0+(?=.)/v, '')}}`, offset);
617
- }
618
818
  const codePoint = Number.parseInt(hex, 16);
819
+ // The value is checked before the spelling, so that the leading zeros error never suggests an escape that is not allowed either.
619
820
  if (codePoint === 0x0D) {
620
- this.#fail(String.raw `A carriage return (U+000D) cannot be represented, so \u{d} is not allowed`, offset);
821
+ this.#fail(String.raw `A carriage return (U+000D) cannot be represented, so \u{${hex}} is not allowed`, offset);
621
822
  }
622
823
  if (codePoint >= 0xD8_00 && codePoint <= 0xDF_FF) {
623
824
  this.#fail(String.raw `\u{${hex}} is a surrogate, which is not a Unicode scalar value`, offset);
@@ -625,10 +826,13 @@ class Parser {
625
826
  if (codePoint > 0x10_FF_FF) {
626
827
  this.#fail(String.raw `\u{${hex}} is above U+10FFFF, the largest Unicode scalar value`, offset);
627
828
  }
829
+ if (hex.length > 1 && hex.startsWith('0')) {
830
+ this.#fail(String.raw `A Unicode escape may not have leading zeros; write \u{${hex.replace(/^0+(?=.)/v, '')}}`, offset);
831
+ }
628
832
  return { text: String.fromCodePoint(codePoint), end: offset + 1 + match[0].length };
629
833
  }
630
834
  // A backslash at the end of the document is the same mistake as one at the end of a line.
631
- const reason = ESCAPE_MISTAKES.get(character ?? '\n') ?? `Unknown escape "\\${String.fromCodePoint(source.codePointAt(offset + 1))}". The escapes are \\\\, \\", \\n, \\t, and \\u{…}; use a '...' string for literal backslashes`;
835
+ const reason = ESCAPE_MISTAKES.get(character ?? '\n') ?? `Unknown escape “\\${String.fromCodePoint(source.codePointAt(offset + 1))}”. The escapes are \\\\, \\", \\n, \\t, and \\u{…}; use a '...' string for literal backslashes`;
632
836
  this.#fail(reason, offset);
633
837
  }
634
838
  /*
@@ -652,7 +856,7 @@ class Parser {
652
856
  const contentStart = this.#index + 1;
653
857
  const closing = findBlockStringEnd(source, contentStart, quote, delimiterLength);
654
858
  if (closing === undefined) {
655
- this.#fail('Unterminated block string', start);
859
+ this.#fail(`Unterminated block string${this.#describeInlineClosingDelimiter(contentStart, quote, delimiterLength)}`, start);
656
860
  }
657
861
  const indentation = source.slice(closing.lineStart, closing.delimiterStart);
658
862
  this.#index = closing.delimiterStart + delimiterLength;
@@ -660,7 +864,7 @@ class Parser {
660
864
  for (let lineStart = contentStart; lineStart < closing.lineStart; lineStart = findLineEnd(source, lineStart) + 1) {
661
865
  const line = source.slice(lineStart, findLineEnd(source, lineStart));
662
866
  // A blank line, which is empty or holds only spaces and tabs, may leave out the indentation, and it becomes an empty line. Swift is stricter here: there only a completely empty line may leave it out.
663
- if (BLANK_LINE.test(line)) {
867
+ if (isBlankLine(line)) {
664
868
  contents.push({ text: '', offset: lineStart });
665
869
  }
666
870
  else if (line.startsWith(indentation)) {
@@ -682,6 +886,24 @@ class Parser {
682
886
  const kept = contents.slice(first, last);
683
887
  return quote === SINGLE_QUOTE ? kept.map(line => line.text).join('\n') : kept.map(line => this.#unescapeLine(line.text, line.offset)).join('\n');
684
888
  }
889
+ /*
890
+ The hint for a content line that ends with the delimiter, as TOML allows. Adding a closing line after it would keep the delimiter as content.
891
+ */
892
+ #describeInlineClosingDelimiter(contentStart, quote, delimiterLength) {
893
+ const source = this.#source;
894
+ const quoteCharacter = String.fromCharCode(quote);
895
+ const delimiter = quoteCharacter.repeat(delimiterLength);
896
+ let line = source.slice(0, contentStart).split('\n').length;
897
+ for (let lineStart = contentStart; lineStart < source.length; lineStart = findLineEnd(source, lineStart) + 1) {
898
+ const text = source.slice(skipSpaces(source, lineStart), skipSpacesBack(source, findLineEnd(source, lineStart)));
899
+ // A line of only quotes, longer than the delimiter, is content.
900
+ if (text.endsWith(delimiter) && text !== quoteCharacter.repeat(text.length)) {
901
+ return `. Its closing delimiter must start a line, so move the ${delimiter} at the end of line ${line} to a new line`;
902
+ }
903
+ line++;
904
+ }
905
+ return '';
906
+ }
685
907
  #unescapeLine(text, offset) {
686
908
  let value = '';
687
909
  let chunkStart = 0;
@@ -710,17 +932,16 @@ class Parser {
710
932
  return this.#parseDuration(text, start);
711
933
  }
712
934
  switch (kind) {
713
- case 'decimal':
714
- case 'radix': {
935
+ case 'int': {
715
936
  if (text === '-0') {
716
- this.#fail('"-0" is not allowed, because zero has one spelling: 0', start);
937
+ this.#fail('“-0” is not allowed, because zero has one spelling: 0', start);
717
938
  }
718
939
  return this.#integer(text.replaceAll('_', ''), text, start);
719
940
  }
720
941
  case 'float': {
721
942
  const value = Number(text.replaceAll('_', ''));
722
943
  if (!Number.isFinite(value)) {
723
- this.#fail(`${abbreviate(text)} is too large to be a finite float. Use infinity if you mean it`, start);
944
+ this.#fail(`${abbreviate(text)} is too large to be a finite float. Use ${value < 0 ? '-infinity' : 'infinity'} if you mean it`, start);
724
945
  }
725
946
  if (value === 0) {
726
947
  // A nonzero digit before the exponent means the literal is not zero, so it underflowed.
@@ -736,7 +957,49 @@ class Parser {
736
957
  break;
737
958
  }
738
959
  }
739
- this.#fail(describeBadNumber(text), start);
960
+ this.#fail(this.#describeThousandsSeparator(text, start) ?? describeBadNumber(text, this.#unquotedText(start)), start);
961
+ }
962
+ /*
963
+ A comma used as a thousands separator, as in `[1,000]`, makes the group after it a separate item, which fails only when it has a leading zero. Writing that item in octal, as the leading zero message suggests, would parse to the wrong items.
964
+ */
965
+ #describeThousandsSeparator(text, start) {
966
+ if (!/^0\d{2}$/v.test(text) || this.#code(start - 1) !== COMMA) {
967
+ return undefined;
968
+ }
969
+ const getGroupStart = (groupEnd) => {
970
+ let groupStart = groupEnd;
971
+ while (isDigit(this.#code(groupStart - 1))) {
972
+ groupStart--;
973
+ }
974
+ return groupStart;
975
+ };
976
+ let groupEnd = start - 1;
977
+ let numberStart = getGroupStart(groupEnd);
978
+ // Earlier groups of three digits, as in `12,345,000`.
979
+ while (groupEnd - numberStart === 3 && this.#code(numberStart - 1) === COMMA && isDigit(this.#code(numberStart - 2))) {
980
+ groupEnd = numberStart - 1;
981
+ numberStart = getGroupStart(groupEnd);
982
+ }
983
+ const before = this.#code(numberStart - 1);
984
+ // The first group has one to three digits, which are not the end of a float, of a number with underscores, or of a number with a radix prefix.
985
+ if (numberStart === groupEnd
986
+ || before === DOT
987
+ || before === 0x5F /* _ */
988
+ || numberStart < groupEnd - 3
989
+ || isLetter(before)) {
990
+ return undefined;
991
+ }
992
+ // The sign belongs to the number, as in `-1,000`.
993
+ if (before === DASH) {
994
+ numberStart--;
995
+ }
996
+ const groupsStart = start + text.length;
997
+ const groups = /^(?:,\d{3}(?!\d))*/v.exec(this.#source.slice(groupsStart, groupsStart + MAX_DIAGNOSED_LENGTH))[0];
998
+ const number = this.#source.slice(numberStart, groupsStart + groups.length);
999
+ const reason = `Leading zeros are not allowed in a decimal number. A comma separates items, so ${abbreviate(number)} is not one number`;
1000
+ // Both spellings are the same int, so they are suggested only when it is in range.
1001
+ const withUnderscores = number.replaceAll(',', '_');
1002
+ return isValidValue(withUnderscores) ? `${reason}. Write it as ${abbreviate(withUnderscores)} or ${abbreviate(number.replaceAll(',', ''))}` : reason;
740
1003
  }
741
1004
  /*
742
1005
  The common case, a short decimal int or float such as `8080`, `-3`, or `30.5`, without the general path's maximal-run scan and classification. Anything else, including every error, returns `undefined` and is left to the general path.
@@ -750,7 +1013,7 @@ class Parser {
750
1013
  index++;
751
1014
  }
752
1015
  const integerLength = index - integerStart;
753
- // No digits, or a leading zero, which is either `0` alone or an error.
1016
+ // No digits, or a leading zero followed by more digits, which the general path reports as an error or reads as an instant before the year 1000.
754
1017
  if (integerLength === 0 || (integerLength > 1 && source.charCodeAt(integerStart) === 0x30)) {
755
1018
  return;
756
1019
  }
@@ -766,9 +1029,8 @@ class Parser {
766
1029
  }
767
1030
  isFloat = true;
768
1031
  }
769
- const next = source.charCodeAt(index);
770
1032
  // At most 15 digits, so an int is exact as a number. A character that may continue a number, such as `e`, `_`, or `-`, needs the general path.
771
- if (index - integerStart > 15 || (index < source.length && (next >= 128 || valueEndCharacters[next] !== 1))) {
1033
+ if (index - integerStart > 15 || !isValueEnd(source.charCodeAt(index))) {
772
1034
  return;
773
1035
  }
774
1036
  const value = Number(source.slice(start, index));
@@ -811,7 +1073,14 @@ class Parser {
811
1073
  #parseInstant(text, start) {
812
1074
  const match = text.length > MAX_DIAGNOSED_LENGTH ? null : INSTANT.exec(text);
813
1075
  if (match === null) {
814
- this.#fail(describeBadInstant(text), start);
1076
+ // A date, a space, and a time, as TOML and Python write a date and time.
1077
+ const timeStart = this.#index + 1;
1078
+ const end = this.#code() === SPACE && isDigit(this.#code(timeStart)) ? findNumberEnd(this.#source, timeStart) : this.#index;
1079
+ const time = this.#source.slice(timeStart, end);
1080
+ // More of the instant may follow, as in `14:00:00,5Z` or `14:00:00[Europe/Oslo]`, so it may have an offset.
1081
+ const next = this.#code(end);
1082
+ const isWholeValue = isValueEnd(next) && !(next === COMMA && isDigit(this.#code(end + 1)));
1083
+ this.#fail(describeBadInstant(text, time, isWholeValue), start);
815
1084
  }
816
1085
  const { year, month, day, hour, minute, second, fraction, offset, offsetHour, offsetMinute } = match.groups;
817
1086
  const check = (isValid, reason) => {
@@ -830,12 +1099,13 @@ class Parser {
830
1099
  if (offset !== 'Z') {
831
1100
  check(offsetHour <= '23', 'the offset hour must be 00 to 23');
832
1101
  check(offsetMinute <= '59', 'the offset minute must be 00 to 59');
833
- check(offset !== '-00:00', '-00:00 means "offset unknown" in RFC 3339, which is not representable; use Z or +00:00');
1102
+ check(offset !== '-00:00', '-00:00 means “offset unknown” in RFC 3339, which is not representable; use Z or +00:00');
834
1103
  }
835
- const instant = requireTemporal().Instant.from(text);
836
- const nanoseconds = instant.epochNanoseconds;
1104
+ // `Date.parse()` reads this format exactly, for every year from 0001 to 9999 and every offset, so the range check needs no `Temporal`.
1105
+ const milliseconds = Date.parse(`${year}-${month}-${day}T${hour}:${minute}:${second}${offset}`);
1106
+ const nanoseconds = (BigInt(milliseconds) * 1000000n) + BigInt((fraction ?? '').padEnd(9, '0'));
837
1107
  check(nanoseconds >= MIN_INSTANT && nanoseconds <= MAX_INSTANT, 'in UTC it falls outside the years 0001 to 9999');
838
- return instant;
1108
+ return this.#createTime(new Time('Instant', nanoseconds));
839
1109
  }
840
1110
  /*
841
1111
  A scan of the parts in order, so that each error names the part it is about.
@@ -859,7 +1129,7 @@ class Parser {
859
1129
  if (code === DASH && isDigit(text.charCodeAt(index + 1))) {
860
1130
  fail('only the whole duration takes a sign, as in -1h30m');
861
1131
  }
862
- fail(code === 0x2B /* + */ ? 'a "+" sign is not allowed' : `expected a number at "${abbreviate(text.slice(index), 10)}"`);
1132
+ fail(code === 0x2B /* + */ ? 'a “+” sign is not allowed' : `expected a number at “${abbreviate(text.slice(index), 10)}”`);
863
1133
  }
864
1134
  if (text.charCodeAt(index) === 0x30 && (next === 0x5F /* _ */ || isDigit(next))) {
865
1135
  fail('leading zeros are not allowed');
@@ -872,7 +1142,7 @@ class Parser {
872
1142
  if (next === DOT) {
873
1143
  unitStart = skipDigits(text, digitsEnd + 1, isDigit);
874
1144
  if (unitStart === digitsEnd + 1) {
875
- fail('a "." must be followed by a digit');
1145
+ fail('a “.” must be followed by a digit');
876
1146
  }
877
1147
  if (text.charCodeAt(unitStart) === 0x5F /* _ */) {
878
1148
  fail('an underscore must be between two digits');
@@ -887,7 +1157,7 @@ class Parser {
887
1157
  const unit = text.slice(unitStart, unitEnd);
888
1158
  const rank = DURATION_UNIT_NAMES.indexOf(unit);
889
1159
  if (rank === -1) {
890
- fail(describeBadDurationUnit(unit));
1160
+ fail(describeBadDurationUnit(unit, text));
891
1161
  }
892
1162
  if (rank <= previousRank) {
893
1163
  fail('the units are in the order h, m, s, ms, us, ns, and each appears at most once');
@@ -915,15 +1185,16 @@ class Parser {
915
1185
  total += fractionNanoseconds / scale;
916
1186
  }
917
1187
  if (isNegative && total === 0n) {
918
- fail('"-" is not allowed before zero, because zero has one spelling: 0s');
1188
+ fail('“-” is not allowed before zero, because zero has one spelling: 0s');
919
1189
  }
920
1190
  if (total > (isNegative ? -INT64_MIN : INT64_MAX)) {
921
1191
  fail('it is outside the 64-bit range of nanoseconds, about 292 years either way');
922
1192
  }
923
- return createDuration(isNegative ? -total : total);
1193
+ return this.#createTime(new Time('Duration', isNegative ? -total : total));
924
1194
  }
925
1195
  parseDocument() {
926
1196
  this.#skipTrivia();
1197
+ this.#documentStart = this.#index;
927
1198
  if (this.#isAtEnd()) {
928
1199
  this.#fail('A document must contain an object or an array, but this one is empty');
929
1200
  }
@@ -946,13 +1217,24 @@ class Parser {
946
1217
  }
947
1218
  }
948
1219
  /*
949
- Whether `text` is a decimal int, a radix int, or a float, following the grammar. A run of digits may have single underscores between digits.
1220
+ Whether `text` is one valid value, so that an error message only suggests a fix that works.
1221
+ */
1222
+ function isValidValue(text) {
1223
+ try {
1224
+ return new Parser(`[${text}]`, 'bigint', time => time).parseDocument().length === 1;
1225
+ }
1226
+ catch {
1227
+ return false;
1228
+ }
1229
+ }
1230
+ /*
1231
+ Whether `text` is an int, in any radix, or a float, following the grammar. A run of digits may have single underscores between digits.
950
1232
  */
951
1233
  function classifyNumber(text) {
952
- const radix = text.charCodeAt(0) === 0x30 ? RADIX_DIGIT[text.charAt(1)] : undefined;
1234
+ const radix = text.charCodeAt(0) === 0x30 ? RADIX_DIGIT.get(text.charAt(1)) : undefined;
953
1235
  if (radix !== undefined) {
954
1236
  const end = skipDigits(text, 2, radix);
955
- return end > 2 && end === text.length ? 'radix' : undefined;
1237
+ return end > 2 && end === text.length ? 'int' : undefined;
956
1238
  }
957
1239
  let index = text.charCodeAt(0) === DASH ? 1 : 0;
958
1240
  // The integer part is `0` or has no leading zero.
@@ -961,7 +1243,7 @@ function classifyNumber(text) {
961
1243
  return undefined;
962
1244
  }
963
1245
  index = integerEnd;
964
- let kind = 'decimal';
1246
+ let kind = 'int';
965
1247
  if (text.charCodeAt(index) === DOT) {
966
1248
  const end = skipDigits(text, index + 1, isDigit);
967
1249
  if (end === index + 1) {
@@ -1037,7 +1319,7 @@ function isDurationLike(text) {
1037
1319
  }
1038
1320
  return /^[dhmnsuw]$/iv.test(text.charAt(index));
1039
1321
  }
1040
- function describeBadDurationUnit(unit) {
1322
+ function describeBadDurationUnit(unit, text) {
1041
1323
  if (unit === '') {
1042
1324
  return 'every number needs a unit: h, m, s, ms, us, or ns';
1043
1325
  }
@@ -1047,12 +1329,24 @@ function describeBadDurationUnit(unit) {
1047
1329
  if (/^w(?:eeks?)?$/v.test(unit)) {
1048
1330
  return 'there is no week unit, because a day is not a fixed length. Write 168h for a fixed 168 hours';
1049
1331
  }
1050
- return DURATION_UNITS.has(unit.toLowerCase()) ? `the units are lowercase: ${unit.toLowerCase()}` : `"${abbreviate(unit, 10)}" is not a unit. The units are h, m, s, ms, us, and ns, and a string must be quoted`;
1332
+ if (unit === 'M') {
1333
+ // A number with `M` alone is more often a size, as in `memory: 512M`, and `512m` would read as minutes.
1334
+ const sizeHint = text.length <= MAX_DIAGNOSED_LENGTH && /^\d[\d_]*M$/v.test(text) ? `. A size, such as 512M, is a string: '${abbreviate(text)}'` : '';
1335
+ return `there is no month unit, because a month is not a fixed length${sizeHint}`;
1336
+ }
1337
+ return DURATION_UNITS.has(unit.toLowerCase()) ? `the units are lowercase: ${unit.toLowerCase()}` : `“${abbreviate(unit, 10)}” is not a unit. The units are h, m, s, ms, us, and ns, and a string must be quoted`;
1338
+ }
1339
+ /*
1340
+ The number of spaces and tabs before `offset` on its line, or `undefined` when something else comes before it.
1341
+ */
1342
+ function getIndentation(source, offset) {
1343
+ const lineStart = source.lastIndexOf('\n', offset - 1) + 1;
1344
+ return skipSpaces(source, lineStart) === offset ? offset - lineStart : undefined;
1051
1345
  }
1052
1346
  function isWordsWithSpaces(text) {
1053
1347
  for (let index = 0; index < text.length; index++) {
1054
1348
  const code = text.charCodeAt(index);
1055
- if (code !== SPACE && code !== TAB && !isBareKeyCharacter(code)) {
1349
+ if (!isSpace(code) && !isBareKeyCharacter(code)) {
1056
1350
  return false;
1057
1351
  }
1058
1352
  }
@@ -1065,23 +1359,38 @@ function daysInMonth(year, month) {
1065
1359
  }
1066
1360
  return [4, 6, 9, 11].includes(month) ? 30 : 31;
1067
1361
  }
1068
- function describeUnknownWord(word) {
1069
- const lowercase = word.toLowerCase();
1362
+ function isUnquotedTextEnd(source, index) {
1363
+ const code = source.charCodeAt(index);
1364
+ const isComment = code === HASH || (code === SLASH && source.charCodeAt(index + 1) === ASTERISK);
1365
+ return code === LF || code === COMMA || code === CLOSE_BRACKET || code === CLOSE_BRACE || (isComment && isSpace(source.charCodeAt(index - 1)));
1366
+ }
1367
+ /*
1368
+ The example in a message that a string must be quoted, which quotes `text` as a `'...'` string. That cannot hold a `'`, so then there is no simple example.
1369
+ */
1370
+ function quotingExample(text) {
1371
+ return text.includes('\'') ? '' : `, as in '${abbreviate(text)}'`;
1372
+ }
1373
+ /*
1374
+ `text` is the unquoted string that `word` begins, which the suggestion quotes.
1375
+ */
1376
+ function describeUnknownWord(word, text) {
1377
+ // A keyword hint is only for the word on its own. In `Yes please` or `Nan Goldin`, following it would leave text behind or change the value.
1378
+ const lowercase = text === word ? word.toLowerCase() : '';
1070
1379
  if (['true', 'false', 'yes', 'no', 'on', 'off'].includes(lowercase)) {
1071
- return `"${word}" is not a value. Booleans are written true and false, in lowercase`;
1380
+ return `“${word}” is not a value. Booleans are written true and false, in lowercase`;
1072
1381
  }
1073
1382
  if (['null', 'nil', 'none', 'undefined'].includes(lowercase)) {
1074
- return `"${word}" is not a value. Null is written null, in lowercase`;
1383
+ return `“${word}” is not a value. Null is written null, in lowercase`;
1075
1384
  }
1076
1385
  if (lowercase === 'nan') {
1077
1386
  return 'NaN is not representable. Use null for a missing value';
1078
1387
  }
1079
- return ['inf', 'infinity'].includes(lowercase) ? `"${word}" is not a value. Infinity is written infinity, in lowercase` : `Unexpected "${abbreviate(word)}". A string value must be quoted, as in '${abbreviate(word)}'`;
1388
+ return ['inf', 'infinity'].includes(lowercase) ? `“${word}” is not a value. Infinity is written infinity, in lowercase` : `Unexpected “${abbreviate(word)}”. A string value must be quoted${quotingExample(text)}`;
1080
1389
  }
1081
1390
  function describeBadUnicodeEscape(source, offset) {
1082
1391
  const rest = source.slice(offset, offset + 12);
1083
1392
  if (/^u[\dA-Fa-f]{4}/v.test(rest)) {
1084
- return `The four-digit \\${rest.slice(0, 5)} form is not an escape. Write \\u{${rest.slice(1, 5).toLowerCase().replace(/^0+(?=.)/v, '')}}`;
1393
+ return describeFourDigitEscape(rest);
1085
1394
  }
1086
1395
  if (/^u\{[\dA-Fa-f]*[A-F]/v.test(rest)) {
1087
1396
  return 'A Unicode escape uses lowercase hexadecimal digits';
@@ -1091,15 +1400,36 @@ function describeBadUnicodeEscape(source, offset) {
1091
1400
  }
1092
1401
  return /^u\{[\da-f]{7}/v.test(rest) ? 'A Unicode escape has at most six hexadecimal digits' : String.raw `A Unicode escape is written \u{…} with one to six lowercase hexadecimal digits`;
1093
1402
  }
1094
- function describeBadNumber(fullText) {
1403
+ /*
1404
+ The JSON form `\uXXXX`, from its `u`, with the escape to write instead. JSON writes a character above U+FFFF as two of them, a surrogate pair, which is one `\u{…}` escape here. Here, a lone surrogate and a carriage return have no escape.
1405
+ */
1406
+ function describeFourDigitEscape(rest) {
1407
+ const code = Number.parseInt(rest.slice(1, 5), 16);
1408
+ const low = /^\\u[\dA-Fa-f]{4}/v.test(rest.slice(5)) ? rest.slice(7, 11) : undefined;
1409
+ const lowCode = low === undefined ? NaN : Number.parseInt(low, 16);
1410
+ if (code >= 0xD8_00 && code <= 0xDB_FF && lowCode >= 0xDC_00 && lowCode <= 0xDF_FF) {
1411
+ const codePoint = 0x1_00_00 + ((code - 0xD8_00) * 0x4_00) + (lowCode - 0xDC_00);
1412
+ return `The four-digit \\${rest.slice(0, 5)}\\u${low} form is not an escape. Write \\u{${codePoint.toString(16)}}`;
1413
+ }
1414
+ const form = `The four-digit \\${rest.slice(0, 5)} form is not an escape`;
1415
+ if (code >= 0xD8_00 && code <= 0xDF_FF) {
1416
+ return String.raw `${form}, and a lone surrogate is not a Unicode scalar value. Write the character it is half of as one \u{…} escape`;
1417
+ }
1418
+ return code === 0x0D ? `${form}, and a carriage return (U+000D) cannot be represented` : String.raw `${form}. Write \u{${code.toString(16)}}`;
1419
+ }
1420
+ /*
1421
+ `unquotedText` is the unquoted string that `fullText` begins, for the suggestion to quote it.
1422
+ */
1423
+ function describeBadNumber(fullText, unquotedText) {
1095
1424
  const text = fullText.slice(0, MAX_DIAGNOSED_LENGTH);
1096
1425
  if (text.includes('+')) {
1097
- return 'A "+" sign is not allowed in a number, including in an exponent';
1426
+ return 'A “+” sign is not allowed in a number, including in an exponent';
1098
1427
  }
1099
1428
  if (/^-?0[BOX]/v.test(text)) {
1100
1429
  return 'A number prefix is lowercase: 0x, 0o, or 0b';
1101
1430
  }
1102
- const radixMatch = /^(?<sign>-?)0(?<radix>[box])(?<digits>.*)$/v.exec(text);
1431
+ // The whole number, not the part cut at the diagnosed length, because a cut can end before the digit or the underscore that the message is about. Each check below takes linear time.
1432
+ const radixMatch = /^(?<sign>-?)0(?<radix>[box])(?<digits>.*)$/v.exec(fullText);
1103
1433
  if (radixMatch !== null) {
1104
1434
  const { sign, radix, digits } = radixMatch.groups;
1105
1435
  const name = RADIX_NAME[radix];
@@ -1107,26 +1437,44 @@ function describeBadNumber(fullText) {
1107
1437
  return `${RADIX_ARTICLE[radix]} ${name} integer cannot have a sign, because it states a bit pattern rather than a quantity`;
1108
1438
  }
1109
1439
  if (digits === '') {
1110
- return `Expected ${name} digits after "0${radix}"`;
1440
+ return `Expected ${name} digits after “0${radix}”`;
1111
1441
  }
1112
1442
  if (radix === 'x' && /[a-f]/v.test(digits) && !/[^\da-f_]/iv.test(digits)) {
1113
- return `Hexadecimal digits are uppercase: 0x${abbreviate(digits.toUpperCase())}`;
1443
+ // The uppercase spelling is only suggested when it is valid, so a misplaced underscore is reported first, and a value outside the 64-bit range gets no example.
1444
+ if (!/^[\da-f]+(?:_[\da-f]+)*$/iv.test(digits)) {
1445
+ return 'An underscore in a number must be between two digits';
1446
+ }
1447
+ const uppercase = `0x${digits.toUpperCase()}`;
1448
+ return isValidValue(uppercase) ? `Hexadecimal digits are uppercase: ${abbreviate(uppercase)}` : 'Hexadecimal digits are uppercase';
1114
1449
  }
1115
- const validCharacter = { x: /[\dA-F_]/v, o: /[0-7_]/v, b: /[01_]/v }[radix];
1450
+ // A lowercase hexadecimal digit is a digit in the wrong case, so the one named is a character that is no digit in either case.
1451
+ const validCharacter = { x: /[\da-f_]/iv, o: /[0-7_]/v, b: /[01_]/v }[radix];
1116
1452
  const character = [...digits].find(character => !validCharacter.test(character));
1117
- return character === undefined ? 'An underscore in a number must be between two digits' : `Invalid ${name} digit "${character}"`;
1453
+ return character === undefined ? 'An underscore in a number must be between two digits' : `Invalid ${name} digit “${character}”`;
1118
1454
  }
1119
1455
  if (/^-?\d[\d_]*E/v.test(text) || /^-?\d[\d_]*\.[\d_]+E/v.test(text)) {
1120
- return 'An exponent marker is a lowercase "e"';
1456
+ return 'An exponent marker is a lowercase “e”';
1121
1457
  }
1122
1458
  if (/^-?[\d_]+\.(?:$|[^\d_])/v.test(text)) {
1123
1459
  return 'A decimal point must be followed by a digit';
1124
1460
  }
1125
1461
  if (text.startsWith('-.')) {
1126
- return 'A number cannot begin with "."; write a digit before it, as in -0.5';
1462
+ return 'A number cannot begin with “.”; write a digit before it, as in -0.5';
1463
+ }
1464
+ if (/^\d+(?::\d+)+$/v.test(text)) {
1465
+ return /^\d{1,2}:\d{2}(?::\d{2}(?:\.\d+)?)?$/v.test(text) ? 'A time of day is a string, so it must be quoted' : `Invalid number “${abbreviate(text)}”. A value that contains “:” must be quoted, as in '${abbreviate(text)}'`;
1127
1466
  }
1128
1467
  if (/^-?0[\d_]/v.test(text)) {
1129
- return 'Leading zeros are not allowed in a decimal number';
1468
+ // Removing the zero gives a valid number with a different meaning, so the message says what the zero usually meant.
1469
+ // A `'...'` string cannot hold a `'`, so then there is no example.
1470
+ const identifier = `an identifier, such as a ZIP code, as a string${unquotedText.includes('\'') ? '' : `: '${abbreviate(unquotedText)}'`}`;
1471
+ // The octal suggestion is only for the number on its own, not for one that more text follows, as in `0412 345 678`, and only when it is in range. It is decided on the whole number, because a cut can end before a digit that is not octal or that changes the value.
1472
+ const digits = fullText.replace(/^0+/v, '');
1473
+ const octal = `0o${digits === '' ? '0' : digits}`;
1474
+ if (unquotedText === fullText && /^0[0-7]+$/v.test(fullText) && isValidValue(octal)) {
1475
+ return `Leading zeros are not allowed in a decimal number. Write an octal number, such as a file mode, as ${abbreviate(octal)}, and ${identifier}`;
1476
+ }
1477
+ return /^0\d+$/v.test(text) ? `Leading zeros are not allowed in a decimal number. Write ${identifier}` : 'Leading zeros are not allowed in a decimal number';
1130
1478
  }
1131
1479
  if (/_(?:$|\D)|(?:^|\D)_/v.test(text)) {
1132
1480
  return 'An underscore in a number must be between two digits';
@@ -1135,37 +1483,65 @@ function describeBadNumber(fullText) {
1135
1483
  return 'Leading zeros are not allowed in an exponent';
1136
1484
  }
1137
1485
  if (/^-?\d[\d_]*(?:\.[\d_]+)?e-0$/v.test(text)) {
1138
- return '"e-0" is not allowed, because an exponent of zero has one spelling: e0';
1486
+ return '“e-0” is not allowed, because an exponent of zero has one spelling: e0';
1139
1487
  }
1140
1488
  if (/e-?$/v.test(text)) {
1141
- return 'Expected digits after the exponent marker "e"';
1489
+ return 'Expected digits after the exponent marker “e”';
1142
1490
  }
1143
1491
  if (text.split('.').length > 2) {
1144
- return `Invalid number "${abbreviate(text)}". A value with several dots, such as a version number, must be quoted`;
1492
+ return `Invalid number “${abbreviate(text)}”. A value with several dots, such as a version number, must be quoted`;
1145
1493
  }
1146
1494
  if (/^-(?:\D|$)/v.test(text)) {
1147
1495
  if (/^-nan/iv.test(text)) {
1148
1496
  return 'NaN is not representable. Use null for a missing value';
1149
1497
  }
1150
- return /^-inf/iv.test(text) ? `"${abbreviate(text)}" is not a value. Negative infinity is written -infinity` : 'Expected a digit or "infinity" after "-"';
1498
+ return /^-inf/iv.test(text) ? `“${abbreviate(text)}” is not a value. Negative infinity is written -infinity` : 'Expected a digit or “infinity” after “-”';
1151
1499
  }
1152
- return /[A-Za-z]/v.test(text) ? `Invalid number "${abbreviate(text)}". A string value must be quoted, as in '${abbreviate(text)}'` : `Invalid number "${abbreviate(text)}"`;
1500
+ return /[A-Za-z]/v.test(text) ? `Invalid number “${abbreviate(text)}”. A string value must be quoted${quotingExample(unquotedText)}` : `Invalid number “${abbreviate(text)}”`;
1153
1501
  }
1154
- function describeBadInstant(text) {
1502
+ /*
1503
+ A date alone, which is not an instant. The instant it could be is only shown when the date exists, so that the example is valid.
1504
+ */
1505
+ function describeDate(date) {
1506
+ const [year, month, day] = date.split('-').map(Number);
1507
+ const isExisting = year >= 1 && month >= 1 && month <= 12 && day >= 1 && day <= daysInMonth(year, month);
1508
+ return `${date} is a date, not an instant. Write a date as a string, as in '${date}'${isExisting ? `. An instant needs a time and an offset, as in ${date}T00:00:00Z` : ''}`;
1509
+ }
1510
+ /*
1511
+ `time` is the token after a space that follows `text`, or an empty string. `isWholeValue` is whether nothing that may be part of the instant follows `text`.
1512
+ */
1513
+ function describeBadInstant(text, time, isWholeValue) {
1155
1514
  if (text.length > MAX_DIAGNOSED_LENGTH) {
1156
- return `Invalid instant "${abbreviate(text)}". An instant is written as 2026-09-19T14:00:00Z, with an optional fraction of up to nine digits and an offset of Z or ±HH:MM`;
1515
+ return `Invalid instant “${abbreviate(text)}”. ${INSTANT_FORMAT}`;
1157
1516
  }
1158
1517
  if (/^\d{4}-\d{2}-\d{2}$/v.test(text)) {
1159
- return `${text} is a date, not an instant. An instant needs a time and an offset, as in ${text}T00:00:00Z; use a string for a date`;
1518
+ return /^\d{2}:\d{2}/v.test(time) ? describeSpaceSeparatedInstant(text, time, isWholeValue) : describeDate(text);
1160
1519
  }
1161
1520
  if (/^\d{4}-\d{2}-\d{2}t/v.test(text)) {
1162
- return 'The date and time separator in an instant is an uppercase "T"';
1521
+ return 'The date and time separator in an instant is an uppercase “T”';
1163
1522
  }
1164
1523
  if (text.endsWith('z')) {
1165
- return 'The UTC offset in an instant is an uppercase "Z"';
1524
+ return 'The UTC offset in an instant is an uppercase “Z”';
1166
1525
  }
1167
1526
  if (/[+\-]\d{4}$/v.test(text)) {
1168
1527
  return 'An instant\'s offset is written with a colon, as in +07:00';
1169
1528
  }
1170
- return /^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d+)?$/v.test(text) ? 'An instant needs an offset: Z or ±HH:MM' : `Invalid instant "${abbreviate(text)}". An instant is written as 2026-09-19T14:00:00Z, with an optional fraction of up to nine digits and an offset of Z or ±HH:MM`;
1529
+ if (!LOCAL_DATE_TIME.test(text)) {
1530
+ return `Invalid instant “${abbreviate(text)}”. ${INSTANT_FORMAT}`;
1531
+ }
1532
+ return isWholeValue ? `An instant needs an offset: Z or ±HH:MM. A date and time without an offset is a local time, which is not an instant, so write it as a string, as in '${abbreviate(text)}', or add the offset it was meant in` : 'An instant needs an offset: Z or ±HH:MM';
1533
+ }
1534
+ /*
1535
+ A date and a time with a space between them, where an instant has a "T". Following the example must not turn a local time into UTC silently, so a time without an offset is described as what it is. An instant is only shown when it is valid, so a date that does not exist, a time out of range, or an instant outside the years 0001 to 9999 in UTC gets the general format instead.
1536
+ */
1537
+ function describeSpaceSeparatedInstant(date, time, isWholeValue) {
1538
+ const separator = 'The date and time separator in an instant is an uppercase “T”, not a space';
1539
+ const instant = `${date}T${time}`;
1540
+ if (instant.length > MAX_DIAGNOSED_LENGTH) {
1541
+ return `${separator}. ${INSTANT_FORMAT}`;
1542
+ }
1543
+ if (isValidValue(instant)) {
1544
+ return `${separator}, as in ${abbreviate(instant)}`;
1545
+ }
1546
+ return isWholeValue && LOCAL_DATE_TIME.test(instant) && isValidValue(`${instant}Z`) ? `${separator}, and an instant needs the offset it was meant in, as in ${abbreviate(instant)}Z for UTC. A date and time without an offset is a local time, which is not an instant, so write it as a string, as in '${abbreviate(`${date} ${time}`)}'` : `${separator}. ${INSTANT_FORMAT}`;
1171
1547
  }