soml-lang 0.0.2 → 0.0.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,43 +1,41 @@
1
1
  import { ParseError } from "./error.js";
2
- import { MAX_DEPTH, INT64_MIN, INT64_MAX, MIN_INSTANT, MAX_INSTANT, DURATION_UNITS, createDuration, requireTemporal, trimTrailingZeros, validateIntegersOption, describeCharacter, formatCodePoint, isBareKey, isBareKeyCharacter, findNumberEnd, findLineEnd, findBlockStringEnd, abbreviate, } from "./shared.js";
3
- const TAB = 0x09;
4
- const LF = 0x0A;
5
- const SPACE = 0x20;
6
- const DOUBLE_QUOTE = 0x22;
7
- const HASH = 0x23;
8
- const SINGLE_QUOTE = 0x27;
9
- const ASTERISK = 0x2A;
10
- const COMMA = 0x2C;
11
- const DASH = 0x2D;
12
- const DOT = 0x2E;
13
- const SLASH = 0x2F;
14
- const COLON = 0x3A;
15
- const OPEN_BRACKET = 0x5B;
16
- const CLOSE_BRACKET = 0x5D;
17
- const OPEN_BRACE = 0x7B;
18
- const CLOSE_BRACE = 0x7D;
19
- /*
20
- What may directly follow a scalar value: whitespace, a separator, a closing bracket, a comment, or the end.
21
- */
2
+ import { formatKey } from "./stringify.js";
3
+ import { MAX_DEPTH, INT64_MIN, INT64_MAX, MIN_INSTANT, MAX_INSTANT, DURATION_UNITS, createDuration, requireTemporal, trimTrailingZeros, getIntegersOption, describeCharacter, formatCodePoint, describeKey, isBareKeyCharacter, isSpace, skipSpaces, skipSpacesBack, isBlankLine, findNumberEnd, findLineEnd, findBlockStringEnd, abbreviate, LF, SPACE, DOUBLE_QUOTE, HASH, SINGLE_QUOTE, ASTERISK, COMMA, DASH, DOT, SLASH, COLON, OPEN_BRACKET, CLOSE_BRACKET, OPEN_BRACE, CLOSE_BRACE, } from "./shared.js";
22
4
  const valueEndCharacters = new Uint8Array(128);
23
5
  for (const character of ' \t\n,]}#/') {
24
6
  valueEndCharacters[character.codePointAt(0)] = 1;
25
7
  }
26
- const RADIX_DIGIT = {
27
- x: code => (code >= 0x30 && code <= 0x39) || (code >= 0x41 && code <= 0x46),
28
- o: code => code >= 0x30 && code <= 0x37,
29
- b: code => code === 0x30 || code === 0x31,
30
- };
31
8
  /*
32
- The longest token the diagnostic regular expressions look at. Their repeated groups need stack in proportion to the input, so a longer invalid token is described from its first part.
9
+ Whether a UTF-16 code unit may directly follow a scalar value: whitespace, a separator, a closing bracket, a comment, or the end, where `charCodeAt()` returns `NaN`.
10
+ */
11
+ function isValueEnd(code) {
12
+ return Number.isNaN(code) || (code < 128 && valueEndCharacters[code] === 1);
13
+ }
14
+ /*
15
+ Whether a UTF-16 code unit is a space, a tab, a line feed, or the end.
16
+ */
17
+ function isSpaceOrLineEnd(code) {
18
+ return isSpace(code) || code === LF || Number.isNaN(code);
19
+ }
20
+ /*
21
+ The lookup tables are maps rather than objects, so that a property added to `Object.prototype` is never read as an entry.
22
+ */
23
+ const RADIX_DIGIT = new Map([
24
+ ['x', code => (code >= 0x30 && code <= 0x39) || (code >= 0x41 && code <= 0x46)],
25
+ ['o', code => code >= 0x30 && code <= 0x37],
26
+ ['b', code => code === 0x30 || code === 0x31],
27
+ ]);
28
+ /*
29
+ The longest token that the instant regular expression and the diagnostic regular expressions look at. Their repeated groups need stack in proportion to the input, so a longer token is rejected with a general message, or described from its first part.
33
30
  */
34
31
  const MAX_DIAGNOSED_LENGTH = 1000;
35
32
  const RADIX_NAME = { x: 'hexadecimal', o: 'octal', b: 'binary' };
36
33
  const RADIX_ARTICLE = { x: 'A', o: 'An', b: 'A' };
37
34
  const INSTANT_PREFIX = /^\d{4}-\d{2}-\d{2}/v;
38
- const INSTANT_START = /^\d{4}-\d{2}-\d{2}T\d{2}:\d/v;
39
35
  const DURATION_UNIT_NAMES = DURATION_UNITS.keys().toArray();
40
36
  const INSTANT = /^(?<year>\d{4})-(?<month>\d{2})-(?<day>\d{2})T(?<hour>\d{2}):(?<minute>\d{2}):(?<second>\d{2})(?:\.(?<fraction>\d+))?(?<offset>Z|[+\-](?<offsetHour>\d{2}):(?<offsetMinute>\d{2}))$/v;
37
+ const LOCAL_DATE_TIME = /^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d+)?$/v;
38
+ const INSTANT_FORMAT = 'An instant is written as 2026-09-19T14:00:00Z, with an optional fraction of up to nine digits and an offset of Z or ±HH:MM';
41
39
  const UNICODE_ESCAPE = /u\{(?<hex>[\da-f]{1,6})\}/vy;
42
40
  /*
43
41
  Every C0 control character except tab and line feed, and DEL. They are errors anywhere in a document.
@@ -45,14 +43,23 @@ Every C0 control character except tab and line feed, and DEL. They are errors an
45
43
  // eslint-disable-next-line no-control-regex, regexp/no-control-character -- Control characters are exactly what must be found.
46
44
  const CONTROL_CHARACTER = /[\u{0}-\u{8}\u{B}-\u{1F}\u{7F}]/v;
47
45
  const LITERAL_STRING_END = /[\n']/gv;
48
- const BLANK_LINE = /^[\t ]*$/v;
46
+ const KEY_QUOTING_HINT = '. A key that contains characters other than letters, digits, “_”, and “-” must be quoted';
47
+ /*
48
+ The characters that may be meant as part of a key: visible ones that have no meaning in the grammar, which includes every bare key character. A `=` is left out, because after a key, as in `name=foo`, it was most likely meant as the `:` of INI and TOML, so a key that holds a `=` gets no quoting hint.
49
+ */
50
+ const QUOTABLE_KEY_CHARACTER = String.raw `[^\p{Default_Ignorable_Code_Point}\p{Other}\p{White_Space}"#'*,.\/:=\[\]\{\}]`;
51
+ const QUOTABLE_KEY_START = new RegExp(`^${QUOTABLE_KEY_CHARACTER}`, 'v');
52
+ /*
53
+ A key made of them, up to its `:`, so the quoting hint can show it quoted.
54
+ */
55
+ const QUOTABLE_KEY = new RegExp(`^${QUOTABLE_KEY_CHARACTER}+(?=:)`, 'v');
49
56
  const ESCAPED_STRING_SPECIAL = /[\n"\\]/gv;
50
- const SIMPLE_ESCAPES = {
51
- '\\': '\\',
52
- '"': '"',
53
- n: '\n',
54
- t: '\t',
55
- };
57
+ const SIMPLE_ESCAPES = new Map([
58
+ ['\\', '\\'],
59
+ ['"', '"'],
60
+ ['n', '\n'],
61
+ ['t', '\t'],
62
+ ]);
56
63
  /*
57
64
  What a backslash followed by one of these characters was probably meant to be.
58
65
  */
@@ -61,13 +68,34 @@ const ESCAPE_MISTAKES = new Map([
61
68
  ['\'', 'A \' needs no escape inside "..."'],
62
69
  ['\n', String.raw `A backslash must be followed by an escape character. Use \\ for a literal backslash, or a '...' string`],
63
70
  ]);
64
- export function parse(text, options = {}) {
65
- const { integers = 'bigint' } = options;
66
- validateIntegersOption(integers);
71
+ export function parse(text, options) {
72
+ const integers = getIntegersOption(options);
67
73
  const source = decode(text);
68
74
  checkCharacters(source);
75
+ // The default `createTime` makes a `Temporal` object of every instant and duration, so no `Time` is left.
69
76
  return new Parser(source, integers).parseDocument();
70
77
  }
78
+ /*
79
+ An instant, as nanoseconds since the Unix epoch, or a duration, as its length in nanoseconds. A class, so that a dotted key sees it as a value and not as an object.
80
+ */
81
+ export class Time {
82
+ type;
83
+ nanoseconds;
84
+ constructor(type, nanoseconds) {
85
+ this.type = type;
86
+ this.nanoseconds = nanoseconds;
87
+ }
88
+ toTemporal() {
89
+ return this.type === 'Instant' ? new (requireTemporal().Instant)(this.nanoseconds) : createDuration(this.nanoseconds);
90
+ }
91
+ }
92
+ /*
93
+ The same as `parse()` for a string, except that each instant and duration is a `Time` instead of a `Temporal` object, so that it works without `Temporal`. For the tree, which makes the `Temporal` object only when the value is read.
94
+ */
95
+ export function parseWithTimes(text) {
96
+ checkCharacters(text);
97
+ return new Parser(text, 'bigint', time => time).parseDocument();
98
+ }
71
99
  const typedArrayTag = Object.getOwnPropertyDescriptor(Object.getPrototypeOf(Uint8Array.prototype), Symbol.toStringTag).get;
72
100
  function decode(text) {
73
101
  if (typeof text === 'string') {
@@ -178,19 +206,25 @@ function describeValue(value) {
178
206
  return value !== null && typeof value === 'object' && Object.getPrototypeOf(value) === Object.prototype ? 'an object' : 'a value';
179
207
  }
180
208
  function formatPath(path) {
181
- return path.map(segment => isBareKey(segment) ? abbreviate(segment) : JSON.stringify(abbreviate(segment))).join('.');
209
+ return path.map(segment => describeKey(segment)).join('.');
182
210
  }
183
211
  class Parser {
184
212
  #source;
185
213
  #index = 0;
186
214
  #integers;
215
+ #createTime;
187
216
  /*
188
217
  Objects created by dotted keys. They may be extended by further dotted keys, while an object written with braces is closed.
189
218
  */
190
219
  #dottedObjects = new WeakSet();
191
- constructor(source, integers) {
220
+ /*
221
+ Where the document's collection starts, after the comments and whitespace before it.
222
+ */
223
+ #documentStart = 0;
224
+ constructor(source, integers, createTime = time => time.toTemporal()) {
192
225
  this.#source = source;
193
226
  this.#integers = integers;
227
+ this.#createTime = createTime;
194
228
  }
195
229
  #fail(reason, offset = this.#index) {
196
230
  throw ParseError.create(reason, this.#source, offset);
@@ -218,11 +252,11 @@ class Parser {
218
252
  }
219
253
  let hasLineBreak = this.#skipTrivia();
220
254
  while (!this.#isAtEnd()) {
255
+ if (this.#code() === COMMA) {
256
+ this.#fail('Top-level entries are separated by line breaks, not commas. Use braces for a one-line object');
257
+ }
221
258
  if (!hasLineBreak) {
222
- if (this.#code() === COMMA) {
223
- this.#fail('Top-level entries are separated by line breaks, not commas. Use braces for a one-line object');
224
- }
225
- this.#fail(`Expected a line break before the next entry, but found ${this.#describeHere()}`);
259
+ this.#fail(`Expected a line break before the next entry, but found ${this.#describeHere()}${this.#slashCommentHint()}`);
226
260
  }
227
261
  this.#parseEntry(object, 1);
228
262
  hasLineBreak = this.#skipTrivia();
@@ -230,21 +264,18 @@ class Parser {
230
264
  return object;
231
265
  }
232
266
  /*
233
- A first line that is one scalar, such as `5` or an instant, fails as an entry. That failure is reported as what it is.
267
+ A whole document that is one scalar, such as `5` or an instant, fails as an entry. That failure is reported as what it is.
234
268
  */
235
269
  #diagnoseBareValue(start) {
236
270
  this.#index = start;
237
271
  let isBareValue = false;
238
272
  try {
239
273
  this.#parseValue(1);
240
- isBareValue = this.#skipTrivia() || this.#isAtEnd();
274
+ this.#skipTrivia();
275
+ isBareValue = this.#isAtEnd();
241
276
  }
242
- catch (error) {
243
- // A bare key cannot hold a ":", so a token shaped like an instant was meant as one, and its own error says what is wrong with it.
244
- if (INSTANT_START.test(this.#source.slice(start, start + 15))) {
245
- throw error;
246
- }
247
- // Otherwise it is not a bare value either, and the original error stands.
277
+ catch {
278
+ // Not a bare value either, so the original error stands.
248
279
  }
249
280
  if (isBareValue) {
250
281
  this.#fail('A bare value is not a document. A document is an object or an array, so write it as `key: value` or `[value]`', start);
@@ -258,7 +289,7 @@ class Parser {
258
289
  let hasCrossedLineBreak = false;
259
290
  for (;;) {
260
291
  const code = source.charCodeAt(this.#index);
261
- if (code === SPACE || code === TAB) {
292
+ if (isSpace(code)) {
262
293
  this.#index++;
263
294
  }
264
295
  else if (code === LF) {
@@ -296,7 +327,7 @@ class Parser {
296
327
  const nested = source.indexOf('/*', start + 2);
297
328
  // The body may not contain `/*`. An opening that overlaps the closing `*/`, as in `/*/`, is not inside the body.
298
329
  if (nested !== -1 && nested + 2 <= end) {
299
- this.#fail('Block comments cannot be nested, and their body may not contain "/*"', nested);
330
+ this.#fail('Block comments cannot be nested, and their body may not contain “/*”', nested);
300
331
  }
301
332
  this.#index = end + 2;
302
333
  }
@@ -307,47 +338,152 @@ class Parser {
307
338
  const keyStart = this.#index;
308
339
  const path = this.#parseKey();
309
340
  if (depth + path.length - 1 > MAX_DEPTH) {
310
- this.#fail(`The document is nested more than ${MAX_DEPTH} levels deep`, keyStart);
341
+ this.#failTooDeep(keyStart);
311
342
  }
312
343
  const code = this.#code();
313
344
  if (code !== COLON) {
314
345
  this.#failMissingColon(keyStart);
315
346
  }
347
+ const colon = this.#index;
316
348
  this.#index++;
317
349
  this.#skipTrivia();
318
- const value = this.#parseValue(depth + path.length);
350
+ let value;
351
+ try {
352
+ value = this.#parseValue(depth + path.length);
353
+ }
354
+ catch (error) {
355
+ if (error instanceof ParseError) {
356
+ this.#diagnoseBadValue(error, path, keyStart, colon);
357
+ }
358
+ throw error;
359
+ }
319
360
  this.#assign(object, path, value, keyStart);
320
361
  }
321
- #failMissingColon(keyStart) {
362
+ /*
363
+ Reports a value that failed to parse as what the entry was meant to be, when that is clear.
364
+ */
365
+ #diagnoseBadValue(error, path, keyStart, colon) {
366
+ // A key that contains a `:`, as in `12:30: 'lunch'`, ends at the first `:`, and the rest is read as the value.
367
+ if (isBareKeyCharacter(this.#code(colon + 1))) {
368
+ this.#diagnoseKeyWithColon(keyStart);
369
+ }
370
+ this.#index = colon + 1;
371
+ const isValueOnNextLine = this.#skipTrivia();
372
+ const valueStart = this.#index;
373
+ const commentHint = this.#describeCommentAsValue(colon);
374
+ if (isValueOnNextLine) {
375
+ this.#diagnoseMissingValue(valueStart, path, keyStart, commentHint);
376
+ }
377
+ // A value that was left out at the end of the document or of an object.
378
+ const code = this.#code(valueStart);
379
+ if (commentHint !== '' && error.offset === valueStart && (code === CLOSE_BRACE || Number.isNaN(code))) {
380
+ this.#fail(`${error.reason}${commentHint}`, valueStart);
381
+ }
382
+ }
383
+ /*
384
+ A bare key followed directly by `:` and more of a key, up to a `:` that ends the key, as in `a:b: 1`.
385
+ */
386
+ #diagnoseKeyWithColon(keyStart) {
322
387
  const source = this.#source;
323
- let next = this.#index;
324
- while (source.charCodeAt(next) === SPACE || source.charCodeAt(next) === TAB) {
325
- next++;
388
+ for (let index = keyStart; index < keyStart + MAX_DIAGNOSED_LENGTH; index++) {
389
+ const code = source.charCodeAt(index);
390
+ if (code === COLON) {
391
+ const next = source.charCodeAt(index + 1);
392
+ if (next === LF || isSpace(next) || Number.isNaN(next)) {
393
+ this.#fail(`A key that contains “:” must be quoted, as in '${abbreviate(source.slice(keyStart, index))}'`, keyStart);
394
+ }
395
+ }
396
+ else if (!isBareKeyCharacter(code)) {
397
+ return;
398
+ }
326
399
  }
400
+ }
401
+ /*
402
+ The hint for a `#` directly after a `:`, as in `color: #FFF`, which starts a comment rather than a value.
403
+ */
404
+ #describeCommentAsValue(colon) {
405
+ const source = this.#source;
406
+ const hash = skipSpaces(source, colon + 1);
407
+ if (source.charCodeAt(hash) !== HASH) {
408
+ return '';
409
+ }
410
+ // A `#` followed by a space, or by another `#`, starts an ordinary comment. A separator or a closing bracket after the value is not part of it. The regular expression, which needs stack in proportion to its match, only sees the start of a long comment, which is cut short in the message anyway.
411
+ const match = /^#[^\t\n #,\]\}][^\t\n ,\]\}]*/v.exec(source.slice(hash, hash + MAX_DIAGNOSED_LENGTH));
412
+ return match === null ? '' : `. “#” starts a comment, so a value that starts with “#” must be quoted${quotingExample(match[0])}`;
413
+ }
414
+ /*
415
+ An entry whose value was left out, as in `a:` followed by `'b': 1` on the next line, reads the next key as the value. That failure is reported as what it is.
416
+ */
417
+ #diagnoseMissingValue(start, path, keyStart, commentHint) {
418
+ // A YAML block sequence.
419
+ if (this.#code(start) === DASH && isSpace(this.#code(start + 1))) {
420
+ this.#fail('Expected a value, but found a “-” list. An array is written in brackets, as in [80, 443]', start);
421
+ }
422
+ this.#index = start;
423
+ let key;
424
+ try {
425
+ key = this.#parseKey();
426
+ }
427
+ catch {
428
+ // Not a key either, so the original error stands.
429
+ }
430
+ if (key === undefined || this.#code() !== COLON) {
431
+ return;
432
+ }
433
+ // Whitespace or the end follows the `:` of a bare key. A digit follows the `:` in an instant such as `2026-09-19T25:00:00Z`, whose own error is more precise. A quoted key cannot be part of a value, so anything may follow its `:`.
434
+ const next = this.#code(this.#index + 1);
435
+ const firstCode = this.#code(start);
436
+ if (firstCode !== SINGLE_QUOTE && firstCode !== DOUBLE_QUOTE && next !== LF && !isSpace(next) && !Number.isNaN(next)) {
437
+ return;
438
+ }
439
+ const hint = commentHint === '' ? this.#describeIndentedKey(path, key, keyStart, start) : commentHint;
440
+ this.#fail(`Expected a value, but found the key ${abbreviate(formatPath(key), 200)}${hint}`, start);
441
+ }
442
+ /*
443
+ The hint for a key at `start` that is indented under the entry whose value is missing, as YAML nests an object.
444
+ */
445
+ #describeIndentedKey(path, key, keyStart, start) {
446
+ const keyIndentation = getIndentation(this.#source, keyStart);
447
+ const indentation = getIndentation(this.#source, start);
448
+ if (keyIndentation === undefined || indentation === undefined || indentation <= keyIndentation) {
449
+ return '';
450
+ }
451
+ // The keys are written as in a document, so that the suggestion is valid, rather than as JSON strings like the rest of the message, whose escapes, such as `\b`, are not all valid.
452
+ const parent = abbreviate(path.map(segment => formatKey(segment)).join('.'), 200);
453
+ const child = abbreviate(key.map(segment => formatKey(segment)).join('.'), 200);
454
+ return `. Indentation does not nest objects, so write ${parent}: {${child}: …} or ${parent}.${child}: …`;
455
+ }
456
+ #failMissingColon(keyStart) {
457
+ const source = this.#source;
458
+ const next = skipSpaces(source, this.#index);
327
459
  const nextCode = source.charCodeAt(next);
328
460
  if (nextCode === COLON && next > this.#index) {
329
- this.#fail('Whitespace is not allowed between a key and its ":"');
461
+ this.#fail('Whitespace is not allowed between a key and its “:”');
330
462
  }
331
- // Nothing else can follow a key, so a block comment there was meant to come before the ":".
463
+ // A block comment followed by the ":" was meant to come before it.
332
464
  if (nextCode === SLASH && source.charCodeAt(next + 1) === ASTERISK) {
333
- this.#fail('A comment is not allowed between a key and its ":"', next);
465
+ const commentEnd = source.indexOf('*/', next + 2);
466
+ if (commentEnd !== -1 && source.charCodeAt(skipSpaces(source, commentEnd + 2)) === COLON) {
467
+ this.#fail('A comment is not allowed between a key and its “:”', next);
468
+ }
334
469
  }
335
470
  const colon = source.indexOf(':', next);
336
- // A key with a space in it, such as `the name: 1`. The hint is only given for plain words, because quoting a dotted or quoted key would change what it means.
337
- if (colon !== -1 && next > this.#index && isBareKeyCharacter(nextCode) && colon < findLineEnd(source, next)) {
338
- const key = source.slice(keyStart, colon).trimEnd();
471
+ // A key with a space in it, such as `the name: 1`. The hint is only given for plain words, because quoting a dotted or quoted key would change what it means, and for a `:` that whitespace or the end follows, because quoting the words before the `:` in `server localhost:8080` would give a valid document with another meaning. So `the name:1` gets no hint.
472
+ if (colon !== -1 && next > this.#index && isBareKeyCharacter(nextCode) && colon < findLineEnd(source, next) && isSpaceOrLineEnd(source.charCodeAt(colon + 1))) {
473
+ // Only spaces and tabs are trimmed, because `trimEnd()` would also remove characters that are not whitespace in SOML, such as U+00A0.
474
+ const key = source.slice(keyStart, skipSpacesBack(source, colon));
339
475
  if (isWordsWithSpaces(key)) {
340
476
  this.#fail(`A bare key cannot contain spaces. Quote it, as in '${abbreviate(key)}'`, keyStart);
341
477
  }
342
478
  }
343
479
  if (this.#isAtEnd() || this.#code() === LF) {
344
- this.#fail('Expected ":" after the key');
480
+ this.#fail('Expected “:” after the key');
345
481
  }
346
482
  // A character directly after a bare key is most likely meant to be part of it.
347
483
  const previousCode = source.charCodeAt(this.#index - 1);
348
- const hint = previousCode !== SINGLE_QUOTE && previousCode !== DOUBLE_QUOTE && next === this.#index ? '. A key that contains characters other than letters, digits, "_", and "-" must be quoted' : '';
484
+ const hint = previousCode !== SINGLE_QUOTE && previousCode !== DOUBLE_QUOTE && next === this.#index && QUOTABLE_KEY_START.test(source.slice(next, next + 2)) ? KEY_QUOTING_HINT : '';
349
485
  this.#index = next;
350
- this.#fail(`Expected ":" after the key, but found ${this.#describeHere()}${hint}`);
486
+ this.#fail(`Expected “:” after the key, but found ${this.#describeHere()}${hint}`);
351
487
  }
352
488
  #parseKey() {
353
489
  const path = [this.#parseKeySegment()];
@@ -373,14 +509,26 @@ class Parser {
373
509
  end++;
374
510
  }
375
511
  if (end === start) {
376
- if (isAfterDot) {
377
- this.#fail(this.#isAtEnd() ? 'Expected a key segment after "."' : `Expected a key segment after ".", but found ${this.#describeHere()}`);
512
+ if (code === OPEN_BRACKET) {
513
+ this.#diagnoseTableHeader();
378
514
  }
379
- this.#fail(this.#isAtEnd() ? 'Expected a key' : `Expected a key, but found ${this.#describeHere()}`);
515
+ const expected = isAfterDot ? 'Expected a key segment after “.”' : 'Expected a key';
516
+ this.#fail(this.#isAtEnd() ? expected : `${expected}, but found ${this.#describeHere()}${this.#keyQuotingHint()}${this.#slashCommentHint()}`);
380
517
  }
381
518
  this.#index = end;
382
519
  return source.slice(start, end);
383
520
  }
521
+ /*
522
+ The hint for a key that starts with a character a bare key cannot hold, such as the `$` in `$schema`, or nothing for a character with another meaning, such as `}`.
523
+ */
524
+ #keyQuotingHint() {
525
+ if (!QUOTABLE_KEY_START.test(this.#source.slice(this.#index, this.#index + 2))) {
526
+ return '';
527
+ }
528
+ // A longer key gets the hint without the example, and the regular expression, which backtracks, never sees a huge run.
529
+ const key = QUOTABLE_KEY.exec(this.#source.slice(this.#index, this.#index + MAX_DIAGNOSED_LENGTH))?.[0];
530
+ return key === undefined ? KEY_QUOTING_HINT : `${KEY_QUOTING_HINT}, as in '${abbreviate(key)}'`;
531
+ }
384
532
  #assign(object, path, value, keyStart) {
385
533
  let target = object;
386
534
  const lastIndex = path.length - 1;
@@ -412,61 +560,59 @@ class Parser {
412
560
  }
413
561
  defineMember(target, key, value);
414
562
  }
563
+ #failTooDeep(offset = this.#index) {
564
+ this.#fail(`The document is nested more than ${MAX_DEPTH} levels deep`, offset);
565
+ }
415
566
  #parseObject(depth) {
416
- if (depth > MAX_DEPTH) {
417
- this.#fail(`The document is nested more than ${MAX_DEPTH} levels deep`);
418
- }
419
- const start = this.#index;
420
567
  const object = {};
421
- this.#index++;
422
- this.#skipTrivia();
423
- for (;;) {
424
- const code = this.#code();
425
- if (code === CLOSE_BRACE) {
426
- this.#index++;
427
- return object;
428
- }
429
- if (this.#isAtEnd()) {
430
- this.#fail('Unterminated object: expected "}"', start);
431
- }
568
+ this.#parseItems(depth, CLOSE_BRACE, () => {
432
569
  this.#parseEntry(object, depth);
433
- this.#skipTrivia();
434
- if (this.#code() === COMMA) {
435
- this.#index++;
436
- this.#skipTrivia();
437
- }
438
- else if (this.#code() !== CLOSE_BRACE) {
439
- this.#fail(this.#isAtEnd() ? 'Unterminated object: expected "}"' : `Expected "," or "}" after an object member, but found ${this.#describeHere()}`, this.#isAtEnd() ? start : this.#index);
440
- }
441
- }
570
+ });
571
+ return object;
442
572
  }
443
573
  #parseArray(depth) {
574
+ const array = [];
575
+ this.#parseItems(depth, CLOSE_BRACKET, () => {
576
+ array.push(this.#parseValue(depth + 1));
577
+ });
578
+ return array;
579
+ }
580
+ /*
581
+ `{` or `[` at depth `depth`, then items separated by a comma, a line break, or both, with an optional trailing comma, then the `closing` bracket. A comma must be on the line of the item before it.
582
+ */
583
+ #parseItems(depth, closing, parseItem) {
444
584
  if (depth > MAX_DEPTH) {
445
- this.#fail(`The document is nested more than ${MAX_DEPTH} levels deep`);
585
+ this.#failTooDeep();
446
586
  }
447
587
  const start = this.#index;
448
- const array = [];
449
588
  this.#index++;
450
589
  this.#skipTrivia();
451
- for (;;) {
452
- const code = this.#code();
453
- if (code === CLOSE_BRACKET) {
454
- this.#index++;
455
- return array;
590
+ while (this.#code() !== closing) {
591
+ let hasLineBreak = false;
592
+ let itemEnd = this.#index;
593
+ if (!this.#isAtEnd()) {
594
+ parseItem();
595
+ itemEnd = this.#index;
596
+ hasLineBreak = this.#skipTrivia();
456
597
  }
457
598
  if (this.#isAtEnd()) {
458
- this.#fail('Unterminated array: expected "]"', start);
599
+ this.#fail(`Unterminated ${closing === CLOSE_BRACE ? 'object' : 'array'}: expected “${String.fromCharCode(closing)}”`, start);
459
600
  }
460
- array.push(this.#parseValue(depth + 1));
461
- this.#skipTrivia();
462
601
  if (this.#code() === COMMA) {
602
+ if (hasLineBreak) {
603
+ this.#fail('A comma must be on the same line as the item before it. The line break already separates the items, so remove the comma');
604
+ }
463
605
  this.#index++;
464
606
  this.#skipTrivia();
465
607
  }
466
- else if (this.#code() !== CLOSE_BRACKET) {
467
- this.#fail(this.#isAtEnd() ? 'Unterminated array: expected "]"' : `Expected "," or "]" after an array item, but found ${this.#describeHere()}`, this.#isAtEnd() ? start : this.#index);
608
+ else if (!hasLineBreak && this.#code() !== closing) {
609
+ const item = closing === CLOSE_BRACE ? 'an object member' : 'an array item';
610
+ // Without a line break that separates, a line break in the gap is inside a block comment.
611
+ const hint = this.#source.slice(itemEnd, this.#index).includes('\n') ? '. A line break inside a block comment does not separate items' : this.#slashCommentHint();
612
+ this.#fail(`Expected “,”, a line break, or “${String.fromCharCode(closing)}” after ${item}, but found ${this.#describeHere()}${hint}`);
468
613
  }
469
614
  }
615
+ this.#index++;
470
616
  }
471
617
  #parseValue(depth) {
472
618
  const code = this.#code();
@@ -476,7 +622,10 @@ class Parser {
476
622
  if (code === OPEN_BRACKET) {
477
623
  return this.#parseArray(depth);
478
624
  }
625
+ const start = this.#index;
479
626
  let value;
627
+ // An int or a float written with digits.
628
+ let isNumber = false;
480
629
  if (code === SINGLE_QUOTE || code === DOUBLE_QUOTE) {
481
630
  value = this.#parseString();
482
631
  }
@@ -497,16 +646,55 @@ class Parser {
497
646
  }
498
647
  else if (code === DASH || isDigit(code)) {
499
648
  value = this.#parseNumberOrInstant();
649
+ isNumber = typeof value === 'bigint' || typeof value === 'number';
500
650
  }
501
651
  else {
502
652
  this.#failUnexpectedValue();
503
653
  }
504
- const next = this.#code();
505
- if (!this.#isAtEnd() && (next >= 128 || valueEndCharacters[next] !== 1)) {
506
- this.#fail(`Unexpected ${this.#describeHere()} after a value`);
654
+ if (!isValueEnd(this.#code())) {
655
+ this.#failAfterValue(start, isNumber);
656
+ }
657
+ if (isNumber && isSpace(this.#code())) {
658
+ this.#diagnoseUnitAfterSpace(start);
507
659
  }
508
660
  return value;
509
661
  }
662
+ /*
663
+ A character that cannot follow the value that starts at `start`.
664
+ */
665
+ #failAfterValue(start, isNumber) {
666
+ const code = this.#code();
667
+ // Go writes microseconds as `µs`, with the micro sign or the Greek letter mu. The duration is only suggested when it is valid, which a number with an exponent or a radix prefix, for example, is not.
668
+ if (isNumber && (code === 0xB5 || code === 0x3_BC) && this.#code(this.#index + 1) === 0x73 /* s */) {
669
+ const number = this.#source.slice(start, this.#index);
670
+ this.#fail(`The unit for microseconds is written us${isValidValue(`${number}us`) ? `, as in ${abbreviate(number)}us` : ''}`);
671
+ }
672
+ // A `''` inside a '...' string, as SQL and YAML escape a quote, ends the string.
673
+ if (code === SINGLE_QUOTE && this.#code(start) === SINGLE_QUOTE) {
674
+ this.#fail('There is no \'\' escape in a \'...\' string. Write a string that contains \' as "...", as in "it\'s"');
675
+ }
676
+ this.#fail(`Unexpected ${this.#describeHere()} after a value`);
677
+ }
678
+ /*
679
+ A number followed by a space and a word, such as `512 MiB` or `10 seconds`, which is an error anyway. A word followed by more than spaces, a separator, or a comment is left to the general errors, because it may be a key, as in `a: 1 b: 2`, or a sentence.
680
+ */
681
+ #diagnoseUnitAfterSpace(start) {
682
+ const source = this.#source;
683
+ const unitStart = skipSpaces(source, this.#index);
684
+ let unitEnd = unitStart;
685
+ while (isLetter(source.charCodeAt(unitEnd))) {
686
+ unitEnd++;
687
+ }
688
+ const unit = source.slice(unitStart, unitEnd);
689
+ // A keyword after a number is a missing comma.
690
+ if (unit === '' || !isValueEnd(source.charCodeAt(skipSpaces(source, unitEnd))) || ['true', 'false', 'null', 'infinity'].includes(unit)) {
691
+ return;
692
+ }
693
+ const number = source.slice(start, this.#index);
694
+ // The duration is only suggested when it is valid, which a number with an exponent or a radix prefix, a fraction of a nanosecond, a negative zero, or a value outside the 64-bit range is not. The string is always offered, because a unit such as `m` may mean meters rather than minutes.
695
+ const duration = DURATION_UNITS.has(unit) && isValidValue(`${number}${unit}`) ? `${abbreviate(number)}${unit}` : '10s';
696
+ this.#fail(`A unit cannot follow a number after a space. Write a duration without the space, as in ${duration}, and anything else, such as a size, as a string, as in '${abbreviate(`${number} ${unit}`)}'`, unitStart);
697
+ }
510
698
  #isKeyword(word) {
511
699
  if (!this.#source.startsWith(word, this.#index)) {
512
700
  return false;
@@ -525,10 +713,11 @@ class Parser {
525
713
  }
526
714
  const code = this.#code();
527
715
  if (code === 0x2B /* + */) {
528
- this.#fail('A "+" sign is not allowed. A number without a sign is positive');
716
+ this.#fail('A “+” sign is not allowed. A number without a sign is positive');
529
717
  }
718
+ // A `.` that no digit follows begins a string, such as `.env` or `./foo`, rather than a number.
530
719
  if (code === DOT) {
531
- this.#fail('A number cannot begin with "."; write a digit before it, as in 0.5');
720
+ this.#fail(isDigit(this.#code(this.#index + 1)) ? 'A number cannot begin with “.”; write a digit before it, as in 0.5' : describeUnknownWord('.', this.#unquotedText()));
532
721
  }
533
722
  let wordEnd = this.#index;
534
723
  while (isBareKeyCharacter(this.#code(wordEnd))) {
@@ -536,13 +725,66 @@ class Parser {
536
725
  }
537
726
  if (wordEnd > this.#index) {
538
727
  const word = this.#source.slice(this.#index, wordEnd);
539
- // An entry whose value was left out, as in `a:` followed by `b: 1` on the next line, reads the next key as the value.
540
- if (this.#code(wordEnd) === COLON) {
728
+ // A key where a value should be, as when the value of an entry is left out and the next line has a key, or as in `[a: 1]`, is reported as a key. A word after the `:` of a member and a space, as in `msg: Error: file not found` or `url: https://example.com`, is that member's value instead, an unquoted string. Without the space, as in `a:b: 1`, the first `:` was most likely meant as part of the key.
729
+ const spacesStart = skipSpacesBack(this.#source, this.#index);
730
+ const isMemberValue = spacesStart < this.#index && this.#code(spacesStart - 1) === COLON;
731
+ if (!isMemberValue && this.#code(wordEnd) === COLON) {
541
732
  this.#fail(`Expected a value, but found the key ${abbreviate(word)}`);
542
733
  }
543
- this.#fail(describeUnknownWord(word));
734
+ this.#diagnoseTableHeader(true);
735
+ this.#fail(describeUnknownWord(word, this.#unquotedText()));
736
+ }
737
+ this.#fail(`Expected a value, but found ${this.#describeHere()}${this.#isBlockScalarIndicator() ? '. Write a multiline string as a block string, between \'\'\' lines' : this.#slashCommentHint()}`);
738
+ }
739
+ /*
740
+ The text from `start` that was most likely meant as one unquoted string, such as `John Smith`: up to the end of the line, a comma, a closing bracket, or a comment after a space or a tab, without the spaces and tabs at its end.
741
+ */
742
+ #unquotedText(start = this.#index) {
743
+ const source = this.#source;
744
+ const limit = Math.min(source.length, start + MAX_DIAGNOSED_LENGTH);
745
+ let end = start;
746
+ while (end < limit && !isUnquotedTextEnd(source, end)) {
747
+ end++;
748
+ }
749
+ return source.slice(start, skipSpacesBack(source, end));
750
+ }
751
+ /*
752
+ A TOML table header, as in `[server]` or `[[servers]]` on a line of its own, reads as an array that holds a word, or as a key that starts with "[". Where a value is expected, `isValue`, it is only a table header when its bracket opens the document, because a table cannot be written as an item of an array inside it.
753
+ */
754
+ #diagnoseTableHeader(isValue = false) {
755
+ const source = this.#source;
756
+ const lineStart = source.lastIndexOf('\n', this.#index - 1) + 1;
757
+ if (isValue && skipSpaces(source, lineStart) !== this.#documentStart) {
758
+ return;
544
759
  }
545
- this.#fail(`Expected a value, but found ${this.#describeHere()}`);
760
+ const line = source.slice(lineStart, Math.min(findLineEnd(source, this.#index), lineStart + MAX_DIAGNOSED_LENGTH));
761
+ // The name is a valid key, so that the suggestions are valid: a dot has a segment on each side.
762
+ const match = /^[\t ]*\[\[?(?<name>[A-Z_a-z][\w\-]*(?:\.[\w\-]+)*)\]\]?[\t ]*$/v.exec(line);
763
+ if (match === null) {
764
+ return;
765
+ }
766
+ const { name } = match.groups;
767
+ this.#fail(`There are no table headers. Write the table as an object, as in ${abbreviate(name)}: {…}, or with dotted keys, as in ${abbreviate(name)}.key: …`);
768
+ }
769
+ /*
770
+ A YAML literal block scalar indicator, as in `key: |` or `key: |-`, at the end of its line.
771
+ */
772
+ #isBlockScalarIndicator() {
773
+ const code = this.#code();
774
+ // A folded block scalar, `>`, joins its lines, which a block string does not, so only `|` gets the hint.
775
+ if (code !== 0x7C /* | */) {
776
+ return false;
777
+ }
778
+ const chomping = this.#code(this.#index + 1);
779
+ const end = skipSpaces(this.#source, chomping === DASH || chomping === 0x2B /* + */ ? this.#index + 2 : this.#index + 1);
780
+ const next = this.#code(end);
781
+ return next === LF || Number.isNaN(next);
782
+ }
783
+ /*
784
+ The hint for a `//` comment, as in JavaScript.
785
+ */
786
+ #slashCommentHint() {
787
+ return this.#code() === SLASH && this.#code(this.#index + 1) === SLASH ? '. A comment starts with “#”' : '';
546
788
  }
547
789
  #describeHere() {
548
790
  if (this.#isAtEnd()) {
@@ -602,8 +844,9 @@ class Parser {
602
844
  #parseEscape(offset) {
603
845
  const source = this.#source;
604
846
  const character = source[offset + 1];
605
- if (character !== undefined && Object.hasOwn(SIMPLE_ESCAPES, character)) {
606
- return { text: SIMPLE_ESCAPES[character], end: offset + 2 };
847
+ const simple = SIMPLE_ESCAPES.get(character ?? '');
848
+ if (simple !== undefined) {
849
+ return { text: simple, end: offset + 2 };
607
850
  }
608
851
  if (character === 'u') {
609
852
  UNICODE_ESCAPE.lastIndex = offset + 1;
@@ -612,12 +855,10 @@ class Parser {
612
855
  this.#fail(describeBadUnicodeEscape(source, offset + 1), offset);
613
856
  }
614
857
  const { hex } = match.groups;
615
- if (hex.length > 1 && hex.startsWith('0')) {
616
- this.#fail(String.raw `A Unicode escape may not have leading zeros; write \u{${hex.replace(/^0+(?=.)/v, '')}}`, offset);
617
- }
618
858
  const codePoint = Number.parseInt(hex, 16);
859
+ // The value is checked before the spelling, so that the leading zeros error never suggests an escape that is not allowed either.
619
860
  if (codePoint === 0x0D) {
620
- this.#fail(String.raw `A carriage return (U+000D) cannot be represented, so \u{d} is not allowed`, offset);
861
+ this.#fail(String.raw `A carriage return (U+000D) cannot be represented, so \u{${hex}} is not allowed`, offset);
621
862
  }
622
863
  if (codePoint >= 0xD8_00 && codePoint <= 0xDF_FF) {
623
864
  this.#fail(String.raw `\u{${hex}} is a surrogate, which is not a Unicode scalar value`, offset);
@@ -625,10 +866,13 @@ class Parser {
625
866
  if (codePoint > 0x10_FF_FF) {
626
867
  this.#fail(String.raw `\u{${hex}} is above U+10FFFF, the largest Unicode scalar value`, offset);
627
868
  }
869
+ if (hex.length > 1 && hex.startsWith('0')) {
870
+ this.#fail(String.raw `A Unicode escape may not have leading zeros; write \u{${hex.replace(/^0+(?=.)/v, '')}}`, offset);
871
+ }
628
872
  return { text: String.fromCodePoint(codePoint), end: offset + 1 + match[0].length };
629
873
  }
630
874
  // A backslash at the end of the document is the same mistake as one at the end of a line.
631
- const reason = ESCAPE_MISTAKES.get(character ?? '\n') ?? `Unknown escape "\\${String.fromCodePoint(source.codePointAt(offset + 1))}". The escapes are \\\\, \\", \\n, \\t, and \\u{…}; use a '...' string for literal backslashes`;
875
+ const reason = ESCAPE_MISTAKES.get(character ?? '\n') ?? `Unknown escape “\\${String.fromCodePoint(source.codePointAt(offset + 1))}”. The escapes are \\\\, \\", \\n, \\t, and \\u{…}; use a '...' string for literal backslashes`;
632
876
  this.#fail(reason, offset);
633
877
  }
634
878
  /*
@@ -652,7 +896,7 @@ class Parser {
652
896
  const contentStart = this.#index + 1;
653
897
  const closing = findBlockStringEnd(source, contentStart, quote, delimiterLength);
654
898
  if (closing === undefined) {
655
- this.#fail('Unterminated block string', start);
899
+ this.#fail(`Unterminated block string${this.#describeInlineClosingDelimiter(contentStart, quote, delimiterLength)}`, start);
656
900
  }
657
901
  const indentation = source.slice(closing.lineStart, closing.delimiterStart);
658
902
  this.#index = closing.delimiterStart + delimiterLength;
@@ -660,7 +904,7 @@ class Parser {
660
904
  for (let lineStart = contentStart; lineStart < closing.lineStart; lineStart = findLineEnd(source, lineStart) + 1) {
661
905
  const line = source.slice(lineStart, findLineEnd(source, lineStart));
662
906
  // A blank line, which is empty or holds only spaces and tabs, may leave out the indentation, and it becomes an empty line. Swift is stricter here: there only a completely empty line may leave it out.
663
- if (BLANK_LINE.test(line)) {
907
+ if (isBlankLine(line)) {
664
908
  contents.push({ text: '', offset: lineStart });
665
909
  }
666
910
  else if (line.startsWith(indentation)) {
@@ -682,6 +926,24 @@ class Parser {
682
926
  const kept = contents.slice(first, last);
683
927
  return quote === SINGLE_QUOTE ? kept.map(line => line.text).join('\n') : kept.map(line => this.#unescapeLine(line.text, line.offset)).join('\n');
684
928
  }
929
+ /*
930
+ The hint for a content line that ends with the delimiter, as TOML allows. Adding a closing line after it would keep the delimiter as content.
931
+ */
932
+ #describeInlineClosingDelimiter(contentStart, quote, delimiterLength) {
933
+ const source = this.#source;
934
+ const quoteCharacter = String.fromCharCode(quote);
935
+ const delimiter = quoteCharacter.repeat(delimiterLength);
936
+ let line = source.slice(0, contentStart).split('\n').length;
937
+ for (let lineStart = contentStart; lineStart < source.length; lineStart = findLineEnd(source, lineStart) + 1) {
938
+ const text = source.slice(skipSpaces(source, lineStart), skipSpacesBack(source, findLineEnd(source, lineStart)));
939
+ // A line of only quotes, longer than the delimiter, is content.
940
+ if (text.endsWith(delimiter) && text !== quoteCharacter.repeat(text.length)) {
941
+ return `. Its closing delimiter must start a line, so move the ${delimiter} at the end of line ${line} to a new line`;
942
+ }
943
+ line++;
944
+ }
945
+ return '';
946
+ }
685
947
  #unescapeLine(text, offset) {
686
948
  let value = '';
687
949
  let chunkStart = 0;
@@ -710,17 +972,16 @@ class Parser {
710
972
  return this.#parseDuration(text, start);
711
973
  }
712
974
  switch (kind) {
713
- case 'decimal':
714
- case 'radix': {
975
+ case 'int': {
715
976
  if (text === '-0') {
716
- this.#fail('"-0" is not allowed, because zero has one spelling: 0', start);
977
+ this.#fail('“-0” is not allowed, because zero has one spelling: 0', start);
717
978
  }
718
979
  return this.#integer(text.replaceAll('_', ''), text, start);
719
980
  }
720
981
  case 'float': {
721
982
  const value = Number(text.replaceAll('_', ''));
722
983
  if (!Number.isFinite(value)) {
723
- this.#fail(`${abbreviate(text)} is too large to be a finite float. Use infinity if you mean it`, start);
984
+ this.#fail(`${abbreviate(text)} is too large to be a finite float. Use ${value < 0 ? '-infinity' : 'infinity'} if you mean it`, start);
724
985
  }
725
986
  if (value === 0) {
726
987
  // A nonzero digit before the exponent means the literal is not zero, so it underflowed.
@@ -736,7 +997,49 @@ class Parser {
736
997
  break;
737
998
  }
738
999
  }
739
- this.#fail(describeBadNumber(text), start);
1000
+ this.#fail(this.#describeThousandsSeparator(text, start) ?? describeBadNumber(text, this.#unquotedText(start)), start);
1001
+ }
1002
+ /*
1003
+ A comma used as a thousands separator, as in `[1,000]`, makes the group after it a separate item, which fails only when it has a leading zero. Writing that item in octal, as the leading zero message suggests, would parse to the wrong items.
1004
+ */
1005
+ #describeThousandsSeparator(text, start) {
1006
+ if (!/^0\d{2}$/v.test(text) || this.#code(start - 1) !== COMMA) {
1007
+ return undefined;
1008
+ }
1009
+ const getGroupStart = (groupEnd) => {
1010
+ let groupStart = groupEnd;
1011
+ while (isDigit(this.#code(groupStart - 1))) {
1012
+ groupStart--;
1013
+ }
1014
+ return groupStart;
1015
+ };
1016
+ let groupEnd = start - 1;
1017
+ let numberStart = getGroupStart(groupEnd);
1018
+ // Earlier groups of three digits, as in `12,345,000`.
1019
+ while (groupEnd - numberStart === 3 && this.#code(numberStart - 1) === COMMA && isDigit(this.#code(numberStart - 2))) {
1020
+ groupEnd = numberStart - 1;
1021
+ numberStart = getGroupStart(groupEnd);
1022
+ }
1023
+ const before = this.#code(numberStart - 1);
1024
+ // The first group has one to three digits, which are not the end of a float, of a number with underscores, or of a number with a radix prefix.
1025
+ if (numberStart === groupEnd
1026
+ || before === DOT
1027
+ || before === 0x5F /* _ */
1028
+ || numberStart < groupEnd - 3
1029
+ || isLetter(before)) {
1030
+ return undefined;
1031
+ }
1032
+ // The sign belongs to the number, as in `-1,000`.
1033
+ if (before === DASH) {
1034
+ numberStart--;
1035
+ }
1036
+ const groupsStart = start + text.length;
1037
+ const groups = /^(?:,\d{3}(?!\d))*/v.exec(this.#source.slice(groupsStart, groupsStart + MAX_DIAGNOSED_LENGTH))[0];
1038
+ const number = this.#source.slice(numberStart, groupsStart + groups.length);
1039
+ const reason = `Leading zeros are not allowed in a decimal number. A comma separates items, so ${abbreviate(number)} is not one number`;
1040
+ // Both spellings are the same int, so they are suggested only when it is in range.
1041
+ const withUnderscores = number.replaceAll(',', '_');
1042
+ return isValidValue(withUnderscores) ? `${reason}. Write it as ${abbreviate(withUnderscores)} or ${abbreviate(number.replaceAll(',', ''))}` : reason;
740
1043
  }
741
1044
  /*
742
1045
  The common case, a short decimal int or float such as `8080`, `-3`, or `30.5`, without the general path's maximal-run scan and classification. Anything else, including every error, returns `undefined` and is left to the general path.
@@ -750,7 +1053,7 @@ class Parser {
750
1053
  index++;
751
1054
  }
752
1055
  const integerLength = index - integerStart;
753
- // No digits, or a leading zero, which is either `0` alone or an error.
1056
+ // No digits, or a leading zero followed by more digits, which the general path reports as an error or reads as an instant before the year 1000.
754
1057
  if (integerLength === 0 || (integerLength > 1 && source.charCodeAt(integerStart) === 0x30)) {
755
1058
  return;
756
1059
  }
@@ -766,9 +1069,8 @@ class Parser {
766
1069
  }
767
1070
  isFloat = true;
768
1071
  }
769
- const next = source.charCodeAt(index);
770
1072
  // At most 15 digits, so an int is exact as a number. A character that may continue a number, such as `e`, `_`, or `-`, needs the general path.
771
- if (index - integerStart > 15 || (index < source.length && (next >= 128 || valueEndCharacters[next] !== 1))) {
1073
+ if (index - integerStart > 15 || !isValueEnd(source.charCodeAt(index))) {
772
1074
  return;
773
1075
  }
774
1076
  const value = Number(source.slice(start, index));
@@ -811,7 +1113,14 @@ class Parser {
811
1113
  #parseInstant(text, start) {
812
1114
  const match = text.length > MAX_DIAGNOSED_LENGTH ? null : INSTANT.exec(text);
813
1115
  if (match === null) {
814
- this.#fail(describeBadInstant(text), start);
1116
+ // A date, a space, and a time, as TOML and Python write a date and time.
1117
+ const timeStart = this.#index + 1;
1118
+ const end = this.#code() === SPACE && isDigit(this.#code(timeStart)) ? findNumberEnd(this.#source, timeStart) : this.#index;
1119
+ const time = this.#source.slice(timeStart, end);
1120
+ // More of the instant may follow, as in `14:00:00,5Z` or `14:00:00[Europe/Oslo]`, so it may have an offset.
1121
+ const next = this.#code(end);
1122
+ const isWholeValue = isValueEnd(next) && !(next === COMMA && isDigit(this.#code(end + 1)));
1123
+ this.#fail(describeBadInstant(text, time, isWholeValue), start);
815
1124
  }
816
1125
  const { year, month, day, hour, minute, second, fraction, offset, offsetHour, offsetMinute } = match.groups;
817
1126
  const check = (isValid, reason) => {
@@ -830,12 +1139,13 @@ class Parser {
830
1139
  if (offset !== 'Z') {
831
1140
  check(offsetHour <= '23', 'the offset hour must be 00 to 23');
832
1141
  check(offsetMinute <= '59', 'the offset minute must be 00 to 59');
833
- check(offset !== '-00:00', '-00:00 means "offset unknown" in RFC 3339, which is not representable; use Z or +00:00');
1142
+ check(offset !== '-00:00', '-00:00 means “offset unknown” in RFC 3339, which is not representable; use Z or +00:00');
834
1143
  }
835
- const instant = requireTemporal().Instant.from(text);
836
- const nanoseconds = instant.epochNanoseconds;
1144
+ // `Date.parse()` reads this format exactly, for every year from 0001 to 9999 and every offset, so the range check needs no `Temporal`.
1145
+ const milliseconds = Date.parse(`${year}-${month}-${day}T${hour}:${minute}:${second}${offset}`);
1146
+ const nanoseconds = (BigInt(milliseconds) * 1000000n) + BigInt((fraction ?? '').padEnd(9, '0'));
837
1147
  check(nanoseconds >= MIN_INSTANT && nanoseconds <= MAX_INSTANT, 'in UTC it falls outside the years 0001 to 9999');
838
- return instant;
1148
+ return this.#createTime(new Time('Instant', nanoseconds));
839
1149
  }
840
1150
  /*
841
1151
  A scan of the parts in order, so that each error names the part it is about.
@@ -859,7 +1169,7 @@ class Parser {
859
1169
  if (code === DASH && isDigit(text.charCodeAt(index + 1))) {
860
1170
  fail('only the whole duration takes a sign, as in -1h30m');
861
1171
  }
862
- fail(code === 0x2B /* + */ ? 'a "+" sign is not allowed' : `expected a number at "${abbreviate(text.slice(index), 10)}"`);
1172
+ fail(code === 0x2B /* + */ ? 'a “+” sign is not allowed' : `expected a number at “${abbreviate(text.slice(index), 10)}”`);
863
1173
  }
864
1174
  if (text.charCodeAt(index) === 0x30 && (next === 0x5F /* _ */ || isDigit(next))) {
865
1175
  fail('leading zeros are not allowed');
@@ -872,7 +1182,7 @@ class Parser {
872
1182
  if (next === DOT) {
873
1183
  unitStart = skipDigits(text, digitsEnd + 1, isDigit);
874
1184
  if (unitStart === digitsEnd + 1) {
875
- fail('a "." must be followed by a digit');
1185
+ fail('a “.” must be followed by a digit');
876
1186
  }
877
1187
  if (text.charCodeAt(unitStart) === 0x5F /* _ */) {
878
1188
  fail('an underscore must be between two digits');
@@ -887,7 +1197,7 @@ class Parser {
887
1197
  const unit = text.slice(unitStart, unitEnd);
888
1198
  const rank = DURATION_UNIT_NAMES.indexOf(unit);
889
1199
  if (rank === -1) {
890
- fail(describeBadDurationUnit(unit));
1200
+ fail(describeBadDurationUnit(unit, text));
891
1201
  }
892
1202
  if (rank <= previousRank) {
893
1203
  fail('the units are in the order h, m, s, ms, us, ns, and each appears at most once');
@@ -915,15 +1225,16 @@ class Parser {
915
1225
  total += fractionNanoseconds / scale;
916
1226
  }
917
1227
  if (isNegative && total === 0n) {
918
- fail('"-" is not allowed before zero, because zero has one spelling: 0s');
1228
+ fail('“-” is not allowed before zero, because zero has one spelling: 0s');
919
1229
  }
920
1230
  if (total > (isNegative ? -INT64_MIN : INT64_MAX)) {
921
1231
  fail('it is outside the 64-bit range of nanoseconds, about 292 years either way');
922
1232
  }
923
- return createDuration(isNegative ? -total : total);
1233
+ return this.#createTime(new Time('Duration', isNegative ? -total : total));
924
1234
  }
925
1235
  parseDocument() {
926
1236
  this.#skipTrivia();
1237
+ this.#documentStart = this.#index;
927
1238
  if (this.#isAtEnd()) {
928
1239
  this.#fail('A document must contain an object or an array, but this one is empty');
929
1240
  }
@@ -946,13 +1257,24 @@ class Parser {
946
1257
  }
947
1258
  }
948
1259
  /*
949
- Whether `text` is a decimal int, a radix int, or a float, following the grammar. A run of digits may have single underscores between digits.
1260
+ Whether `text` is one valid value, so that an error message only suggests a fix that works.
1261
+ */
1262
+ function isValidValue(text) {
1263
+ try {
1264
+ return new Parser(`[${text}]`, 'bigint', time => time).parseDocument().length === 1;
1265
+ }
1266
+ catch {
1267
+ return false;
1268
+ }
1269
+ }
1270
+ /*
1271
+ Whether `text` is an int, in any radix, or a float, following the grammar. A run of digits may have single underscores between digits.
950
1272
  */
951
1273
  function classifyNumber(text) {
952
- const radix = text.charCodeAt(0) === 0x30 ? RADIX_DIGIT[text.charAt(1)] : undefined;
1274
+ const radix = text.charCodeAt(0) === 0x30 ? RADIX_DIGIT.get(text.charAt(1)) : undefined;
953
1275
  if (radix !== undefined) {
954
1276
  const end = skipDigits(text, 2, radix);
955
- return end > 2 && end === text.length ? 'radix' : undefined;
1277
+ return end > 2 && end === text.length ? 'int' : undefined;
956
1278
  }
957
1279
  let index = text.charCodeAt(0) === DASH ? 1 : 0;
958
1280
  // The integer part is `0` or has no leading zero.
@@ -961,7 +1283,7 @@ function classifyNumber(text) {
961
1283
  return undefined;
962
1284
  }
963
1285
  index = integerEnd;
964
- let kind = 'decimal';
1286
+ let kind = 'int';
965
1287
  if (text.charCodeAt(index) === DOT) {
966
1288
  const end = skipDigits(text, index + 1, isDigit);
967
1289
  if (end === index + 1) {
@@ -1037,7 +1359,7 @@ function isDurationLike(text) {
1037
1359
  }
1038
1360
  return /^[dhmnsuw]$/iv.test(text.charAt(index));
1039
1361
  }
1040
- function describeBadDurationUnit(unit) {
1362
+ function describeBadDurationUnit(unit, text) {
1041
1363
  if (unit === '') {
1042
1364
  return 'every number needs a unit: h, m, s, ms, us, or ns';
1043
1365
  }
@@ -1047,12 +1369,24 @@ function describeBadDurationUnit(unit) {
1047
1369
  if (/^w(?:eeks?)?$/v.test(unit)) {
1048
1370
  return 'there is no week unit, because a day is not a fixed length. Write 168h for a fixed 168 hours';
1049
1371
  }
1050
- return DURATION_UNITS.has(unit.toLowerCase()) ? `the units are lowercase: ${unit.toLowerCase()}` : `"${abbreviate(unit, 10)}" is not a unit. The units are h, m, s, ms, us, and ns, and a string must be quoted`;
1372
+ if (unit === 'M') {
1373
+ // A number with `M` alone is more often a size, as in `memory: 512M`, and `512m` would read as minutes.
1374
+ const sizeHint = text.length <= MAX_DIAGNOSED_LENGTH && /^\d[\d_]*M$/v.test(text) ? `. A size, such as 512M, is a string: '${abbreviate(text)}'` : '';
1375
+ return `there is no month unit, because a month is not a fixed length${sizeHint}`;
1376
+ }
1377
+ return DURATION_UNITS.has(unit.toLowerCase()) ? `the units are lowercase: ${unit.toLowerCase()}` : `“${abbreviate(unit, 10)}” is not a unit. The units are h, m, s, ms, us, and ns, and a string must be quoted`;
1378
+ }
1379
+ /*
1380
+ The number of spaces and tabs before `offset` on its line, or `undefined` when something else comes before it.
1381
+ */
1382
+ function getIndentation(source, offset) {
1383
+ const lineStart = source.lastIndexOf('\n', offset - 1) + 1;
1384
+ return skipSpaces(source, lineStart) === offset ? offset - lineStart : undefined;
1051
1385
  }
1052
1386
  function isWordsWithSpaces(text) {
1053
1387
  for (let index = 0; index < text.length; index++) {
1054
1388
  const code = text.charCodeAt(index);
1055
- if (code !== SPACE && code !== TAB && !isBareKeyCharacter(code)) {
1389
+ if (!isSpace(code) && !isBareKeyCharacter(code)) {
1056
1390
  return false;
1057
1391
  }
1058
1392
  }
@@ -1065,23 +1399,38 @@ function daysInMonth(year, month) {
1065
1399
  }
1066
1400
  return [4, 6, 9, 11].includes(month) ? 30 : 31;
1067
1401
  }
1068
- function describeUnknownWord(word) {
1069
- const lowercase = word.toLowerCase();
1402
+ function isUnquotedTextEnd(source, index) {
1403
+ const code = source.charCodeAt(index);
1404
+ const isComment = code === HASH || (code === SLASH && source.charCodeAt(index + 1) === ASTERISK);
1405
+ return code === LF || code === COMMA || code === CLOSE_BRACKET || code === CLOSE_BRACE || (isComment && isSpace(source.charCodeAt(index - 1)));
1406
+ }
1407
+ /*
1408
+ The example in a message that a string must be quoted, which quotes `text` as a `'...'` string. That cannot hold a `'`, so then there is no simple example.
1409
+ */
1410
+ function quotingExample(text) {
1411
+ return text.includes('\'') ? '' : `, as in '${abbreviate(text)}'`;
1412
+ }
1413
+ /*
1414
+ `text` is the unquoted string that `word` begins, which the suggestion quotes.
1415
+ */
1416
+ function describeUnknownWord(word, text) {
1417
+ // A keyword hint is only for the word on its own. In `Yes please` or `Nan Goldin`, following it would leave text behind or change the value.
1418
+ const lowercase = text === word ? word.toLowerCase() : '';
1070
1419
  if (['true', 'false', 'yes', 'no', 'on', 'off'].includes(lowercase)) {
1071
- return `"${word}" is not a value. Booleans are written true and false, in lowercase`;
1420
+ return `“${word}” is not a value. Booleans are written true and false, in lowercase`;
1072
1421
  }
1073
1422
  if (['null', 'nil', 'none', 'undefined'].includes(lowercase)) {
1074
- return `"${word}" is not a value. Null is written null, in lowercase`;
1423
+ return `“${word}” is not a value. Null is written null, in lowercase`;
1075
1424
  }
1076
1425
  if (lowercase === 'nan') {
1077
1426
  return 'NaN is not representable. Use null for a missing value';
1078
1427
  }
1079
- return ['inf', 'infinity'].includes(lowercase) ? `"${word}" is not a value. Infinity is written infinity, in lowercase` : `Unexpected "${abbreviate(word)}". A string value must be quoted, as in '${abbreviate(word)}'`;
1428
+ return ['inf', 'infinity'].includes(lowercase) ? `“${word}” is not a value. Infinity is written infinity, in lowercase` : `Unexpected “${abbreviate(word)}”. A string value must be quoted${quotingExample(text)}`;
1080
1429
  }
1081
1430
  function describeBadUnicodeEscape(source, offset) {
1082
1431
  const rest = source.slice(offset, offset + 12);
1083
1432
  if (/^u[\dA-Fa-f]{4}/v.test(rest)) {
1084
- return `The four-digit \\${rest.slice(0, 5)} form is not an escape. Write \\u{${rest.slice(1, 5).toLowerCase().replace(/^0+(?=.)/v, '')}}`;
1433
+ return describeFourDigitEscape(rest);
1085
1434
  }
1086
1435
  if (/^u\{[\dA-Fa-f]*[A-F]/v.test(rest)) {
1087
1436
  return 'A Unicode escape uses lowercase hexadecimal digits';
@@ -1091,15 +1440,36 @@ function describeBadUnicodeEscape(source, offset) {
1091
1440
  }
1092
1441
  return /^u\{[\da-f]{7}/v.test(rest) ? 'A Unicode escape has at most six hexadecimal digits' : String.raw `A Unicode escape is written \u{…} with one to six lowercase hexadecimal digits`;
1093
1442
  }
1094
- function describeBadNumber(fullText) {
1443
+ /*
1444
+ The JSON form `\uXXXX`, from its `u`, with the escape to write instead. JSON writes a character above U+FFFF as two of them, a surrogate pair, which is one `\u{…}` escape here. Here, a lone surrogate and a carriage return have no escape.
1445
+ */
1446
+ function describeFourDigitEscape(rest) {
1447
+ const code = Number.parseInt(rest.slice(1, 5), 16);
1448
+ const low = /^\\u[\dA-Fa-f]{4}/v.test(rest.slice(5)) ? rest.slice(7, 11) : undefined;
1449
+ const lowCode = low === undefined ? NaN : Number.parseInt(low, 16);
1450
+ if (code >= 0xD8_00 && code <= 0xDB_FF && lowCode >= 0xDC_00 && lowCode <= 0xDF_FF) {
1451
+ const codePoint = 0x1_00_00 + ((code - 0xD8_00) * 0x4_00) + (lowCode - 0xDC_00);
1452
+ return `The four-digit \\${rest.slice(0, 5)}\\u${low} form is not an escape. Write \\u{${codePoint.toString(16)}}`;
1453
+ }
1454
+ const form = `The four-digit \\${rest.slice(0, 5)} form is not an escape`;
1455
+ if (code >= 0xD8_00 && code <= 0xDF_FF) {
1456
+ return String.raw `${form}, and a lone surrogate is not a Unicode scalar value. Write the character it is half of as one \u{…} escape`;
1457
+ }
1458
+ return code === 0x0D ? `${form}, and a carriage return (U+000D) cannot be represented` : String.raw `${form}. Write \u{${code.toString(16)}}`;
1459
+ }
1460
+ /*
1461
+ `unquotedText` is the unquoted string that `fullText` begins, for the suggestion to quote it.
1462
+ */
1463
+ function describeBadNumber(fullText, unquotedText) {
1095
1464
  const text = fullText.slice(0, MAX_DIAGNOSED_LENGTH);
1096
1465
  if (text.includes('+')) {
1097
- return 'A "+" sign is not allowed in a number, including in an exponent';
1466
+ return 'A “+” sign is not allowed in a number, including in an exponent';
1098
1467
  }
1099
1468
  if (/^-?0[BOX]/v.test(text)) {
1100
1469
  return 'A number prefix is lowercase: 0x, 0o, or 0b';
1101
1470
  }
1102
- const radixMatch = /^(?<sign>-?)0(?<radix>[box])(?<digits>.*)$/v.exec(text);
1471
+ // The whole number, not the part cut at the diagnosed length, because a cut can end before the digit or the underscore that the message is about. Each check below takes linear time.
1472
+ const radixMatch = /^(?<sign>-?)0(?<radix>[box])(?<digits>.*)$/v.exec(fullText);
1103
1473
  if (radixMatch !== null) {
1104
1474
  const { sign, radix, digits } = radixMatch.groups;
1105
1475
  const name = RADIX_NAME[radix];
@@ -1107,26 +1477,44 @@ function describeBadNumber(fullText) {
1107
1477
  return `${RADIX_ARTICLE[radix]} ${name} integer cannot have a sign, because it states a bit pattern rather than a quantity`;
1108
1478
  }
1109
1479
  if (digits === '') {
1110
- return `Expected ${name} digits after "0${radix}"`;
1480
+ return `Expected ${name} digits after “0${radix}”`;
1111
1481
  }
1112
1482
  if (radix === 'x' && /[a-f]/v.test(digits) && !/[^\da-f_]/iv.test(digits)) {
1113
- return `Hexadecimal digits are uppercase: 0x${abbreviate(digits.toUpperCase())}`;
1483
+ // The uppercase spelling is only suggested when it is valid, so a misplaced underscore is reported first, and a value outside the 64-bit range gets no example.
1484
+ if (!/^[\da-f]+(?:_[\da-f]+)*$/iv.test(digits)) {
1485
+ return 'An underscore in a number must be between two digits';
1486
+ }
1487
+ const uppercase = `0x${digits.toUpperCase()}`;
1488
+ return isValidValue(uppercase) ? `Hexadecimal digits are uppercase: ${abbreviate(uppercase)}` : 'Hexadecimal digits are uppercase';
1114
1489
  }
1115
- const validCharacter = { x: /[\dA-F_]/v, o: /[0-7_]/v, b: /[01_]/v }[radix];
1490
+ // A lowercase hexadecimal digit is a digit in the wrong case, so the one named is a character that is no digit in either case.
1491
+ const validCharacter = { x: /[\da-f_]/iv, o: /[0-7_]/v, b: /[01_]/v }[radix];
1116
1492
  const character = [...digits].find(character => !validCharacter.test(character));
1117
- return character === undefined ? 'An underscore in a number must be between two digits' : `Invalid ${name} digit "${character}"`;
1493
+ return character === undefined ? 'An underscore in a number must be between two digits' : `Invalid ${name} digit “${character}”`;
1118
1494
  }
1119
1495
  if (/^-?\d[\d_]*E/v.test(text) || /^-?\d[\d_]*\.[\d_]+E/v.test(text)) {
1120
- return 'An exponent marker is a lowercase "e"';
1496
+ return 'An exponent marker is a lowercase “e”';
1121
1497
  }
1122
1498
  if (/^-?[\d_]+\.(?:$|[^\d_])/v.test(text)) {
1123
1499
  return 'A decimal point must be followed by a digit';
1124
1500
  }
1125
1501
  if (text.startsWith('-.')) {
1126
- return 'A number cannot begin with "."; write a digit before it, as in -0.5';
1502
+ return 'A number cannot begin with “.”; write a digit before it, as in -0.5';
1503
+ }
1504
+ if (/^\d+(?::\d+)+$/v.test(text)) {
1505
+ return /^\d{1,2}:\d{2}(?::\d{2}(?:\.\d+)?)?$/v.test(text) ? 'A time of day is a string, so it must be quoted' : `Invalid number “${abbreviate(text)}”. A value that contains “:” must be quoted, as in '${abbreviate(text)}'`;
1127
1506
  }
1128
1507
  if (/^-?0[\d_]/v.test(text)) {
1129
- return 'Leading zeros are not allowed in a decimal number';
1508
+ // Removing the zero gives a valid number with a different meaning, so the message says what the zero usually meant.
1509
+ // A `'...'` string cannot hold a `'`, so then there is no example.
1510
+ const identifier = `an identifier, such as a ZIP code, as a string${unquotedText.includes('\'') ? '' : `: '${abbreviate(unquotedText)}'`}`;
1511
+ // The octal suggestion is only for the number on its own, not for one that more text follows, as in `0412 345 678`, and only when it is in range. It is decided on the whole number, because a cut can end before a digit that is not octal or that changes the value.
1512
+ const digits = fullText.replace(/^0+/v, '');
1513
+ const octal = `0o${digits === '' ? '0' : digits}`;
1514
+ if (unquotedText === fullText && /^0[0-7]+$/v.test(fullText) && isValidValue(octal)) {
1515
+ return `Leading zeros are not allowed in a decimal number. Write an octal number, such as a file mode, as ${abbreviate(octal)}, and ${identifier}`;
1516
+ }
1517
+ return /^0\d+$/v.test(text) ? `Leading zeros are not allowed in a decimal number. Write ${identifier}` : 'Leading zeros are not allowed in a decimal number';
1130
1518
  }
1131
1519
  if (/_(?:$|\D)|(?:^|\D)_/v.test(text)) {
1132
1520
  return 'An underscore in a number must be between two digits';
@@ -1135,37 +1523,65 @@ function describeBadNumber(fullText) {
1135
1523
  return 'Leading zeros are not allowed in an exponent';
1136
1524
  }
1137
1525
  if (/^-?\d[\d_]*(?:\.[\d_]+)?e-0$/v.test(text)) {
1138
- return '"e-0" is not allowed, because an exponent of zero has one spelling: e0';
1526
+ return '“e-0” is not allowed, because an exponent of zero has one spelling: e0';
1139
1527
  }
1140
1528
  if (/e-?$/v.test(text)) {
1141
- return 'Expected digits after the exponent marker "e"';
1529
+ return 'Expected digits after the exponent marker “e”';
1142
1530
  }
1143
1531
  if (text.split('.').length > 2) {
1144
- return `Invalid number "${abbreviate(text)}". A value with several dots, such as a version number, must be quoted`;
1532
+ return `Invalid number “${abbreviate(text)}”. A value with several dots, such as a version number, must be quoted`;
1145
1533
  }
1146
1534
  if (/^-(?:\D|$)/v.test(text)) {
1147
1535
  if (/^-nan/iv.test(text)) {
1148
1536
  return 'NaN is not representable. Use null for a missing value';
1149
1537
  }
1150
- return /^-inf/iv.test(text) ? `"${abbreviate(text)}" is not a value. Negative infinity is written -infinity` : 'Expected a digit or "infinity" after "-"';
1538
+ return /^-inf/iv.test(text) ? `“${abbreviate(text)}” is not a value. Negative infinity is written -infinity` : 'Expected a digit or “infinity” after “-”';
1151
1539
  }
1152
- return /[A-Za-z]/v.test(text) ? `Invalid number "${abbreviate(text)}". A string value must be quoted, as in '${abbreviate(text)}'` : `Invalid number "${abbreviate(text)}"`;
1540
+ return /[A-Za-z]/v.test(text) ? `Invalid number “${abbreviate(text)}”. A string value must be quoted${quotingExample(unquotedText)}` : `Invalid number “${abbreviate(text)}”`;
1153
1541
  }
1154
- function describeBadInstant(text) {
1542
+ /*
1543
+ A date alone, which is not an instant. The instant it could be is only shown when the date exists, so that the example is valid.
1544
+ */
1545
+ function describeDate(date) {
1546
+ const [year, month, day] = date.split('-').map(Number);
1547
+ const isExisting = year >= 1 && month >= 1 && month <= 12 && day >= 1 && day <= daysInMonth(year, month);
1548
+ return `${date} is a date, not an instant. Write a date as a string, as in '${date}'${isExisting ? `. An instant needs a time and an offset, as in ${date}T00:00:00Z` : ''}`;
1549
+ }
1550
+ /*
1551
+ `time` is the token after a space that follows `text`, or an empty string. `isWholeValue` is whether nothing that may be part of the instant follows `text`.
1552
+ */
1553
+ function describeBadInstant(text, time, isWholeValue) {
1155
1554
  if (text.length > MAX_DIAGNOSED_LENGTH) {
1156
- return `Invalid instant "${abbreviate(text)}". An instant is written as 2026-09-19T14:00:00Z, with an optional fraction of up to nine digits and an offset of Z or ±HH:MM`;
1555
+ return `Invalid instant “${abbreviate(text)}”. ${INSTANT_FORMAT}`;
1157
1556
  }
1158
1557
  if (/^\d{4}-\d{2}-\d{2}$/v.test(text)) {
1159
- return `${text} is a date, not an instant. An instant needs a time and an offset, as in ${text}T00:00:00Z; use a string for a date`;
1558
+ return /^\d{2}:\d{2}/v.test(time) ? describeSpaceSeparatedInstant(text, time, isWholeValue) : describeDate(text);
1160
1559
  }
1161
1560
  if (/^\d{4}-\d{2}-\d{2}t/v.test(text)) {
1162
- return 'The date and time separator in an instant is an uppercase "T"';
1561
+ return 'The date and time separator in an instant is an uppercase “T”';
1163
1562
  }
1164
1563
  if (text.endsWith('z')) {
1165
- return 'The UTC offset in an instant is an uppercase "Z"';
1564
+ return 'The UTC offset in an instant is an uppercase “Z”';
1166
1565
  }
1167
1566
  if (/[+\-]\d{4}$/v.test(text)) {
1168
1567
  return 'An instant\'s offset is written with a colon, as in +07:00';
1169
1568
  }
1170
- return /^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d+)?$/v.test(text) ? 'An instant needs an offset: Z or ±HH:MM' : `Invalid instant "${abbreviate(text)}". An instant is written as 2026-09-19T14:00:00Z, with an optional fraction of up to nine digits and an offset of Z or ±HH:MM`;
1569
+ if (!LOCAL_DATE_TIME.test(text)) {
1570
+ return `Invalid instant “${abbreviate(text)}”. ${INSTANT_FORMAT}`;
1571
+ }
1572
+ return isWholeValue ? `An instant needs an offset: Z or ±HH:MM. A date and time without an offset is a local time, which is not an instant, so write it as a string, as in '${abbreviate(text)}', or add the offset it was meant in` : 'An instant needs an offset: Z or ±HH:MM';
1573
+ }
1574
+ /*
1575
+ A date and a time with a space between them, where an instant has a "T". Following the example must not turn a local time into UTC silently, so a time without an offset is described as what it is. An instant is only shown when it is valid, so a date that does not exist, a time out of range, or an instant outside the years 0001 to 9999 in UTC gets the general format instead.
1576
+ */
1577
+ function describeSpaceSeparatedInstant(date, time, isWholeValue) {
1578
+ const separator = 'The date and time separator in an instant is an uppercase “T”, not a space';
1579
+ const instant = `${date}T${time}`;
1580
+ if (instant.length > MAX_DIAGNOSED_LENGTH) {
1581
+ return `${separator}. ${INSTANT_FORMAT}`;
1582
+ }
1583
+ if (isValidValue(instant)) {
1584
+ return `${separator}, as in ${abbreviate(instant)}`;
1585
+ }
1586
+ return isWholeValue && LOCAL_DATE_TIME.test(instant) && isValidValue(`${instant}Z`) ? `${separator}, and an instant needs the offset it was meant in, as in ${abbreviate(instant)}Z for UTC. A date and time without an offset is a local time, which is not an instant, so write it as a string, as in '${abbreviate(`${date} ${time}`)}'` : `${separator}. ${INSTANT_FORMAT}`;
1171
1587
  }