aontu 0.56.0 → 0.57.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (69) hide show
  1. package/dist/aontu.d.ts +2 -1
  2. package/dist/aontu.js +1 -1
  3. package/dist/aontu.js.map +1 -1
  4. package/dist/cli.d.ts +5 -1
  5. package/dist/cli.js +89 -18
  6. package/dist/cli.js.map +1 -1
  7. package/dist/escape.d.ts +5 -0
  8. package/dist/escape.js +455 -0
  9. package/dist/escape.js.map +1 -0
  10. package/dist/format.d.ts +9 -0
  11. package/dist/format.js +544 -55
  12. package/dist/format.js.map +1 -1
  13. package/dist/hints.js +35 -1
  14. package/dist/hints.js.map +1 -1
  15. package/dist/lang.js +20 -1
  16. package/dist/lang.js.map +1 -1
  17. package/dist/lsp.d.ts +1 -1
  18. package/dist/lsp.js +3 -3
  19. package/dist/lsp.js.map +1 -1
  20. package/dist/mcp-server.js +2 -2
  21. package/dist/mcp-server.js.map +1 -1
  22. package/dist/mod-tool.js +8 -7
  23. package/dist/mod-tool.js.map +1 -1
  24. package/dist/mod.js +6 -6
  25. package/dist/mod.js.map +1 -1
  26. package/dist/sigdecl.js +1 -1
  27. package/dist/sigdecl.js.map +1 -1
  28. package/dist/tsconfig.tsbuildinfo +1 -1
  29. package/dist/val/EachFuncVal.js +1 -2
  30. package/dist/val/EachFuncVal.js.map +1 -1
  31. package/dist/val/EmitFuncVal.d.ts +23 -0
  32. package/dist/val/EmitFuncVal.js +261 -0
  33. package/dist/val/EmitFuncVal.js.map +1 -0
  34. package/dist/val/FilterFuncVal.js +1 -2
  35. package/dist/val/FilterFuncVal.js.map +1 -1
  36. package/dist/val/FuncBaseVal.d.ts +1 -0
  37. package/dist/val/FuncBaseVal.js +16 -0
  38. package/dist/val/FuncBaseVal.js.map +1 -1
  39. package/dist/val/PackFuncVal.js +1 -2
  40. package/dist/val/PackFuncVal.js.map +1 -1
  41. package/dist/val/PlaceVal.d.ts +3 -1
  42. package/dist/val/PlaceVal.js +8 -6
  43. package/dist/val/PlaceVal.js.map +1 -1
  44. package/dist/val/StrFuncVal.d.ts +32 -0
  45. package/dist/val/StrFuncVal.js +292 -0
  46. package/dist/val/StrFuncVal.js.map +1 -0
  47. package/grammar/aontu.abnf +4 -3
  48. package/grammar/aontu.gbnf +4 -3
  49. package/grammar/aontu.lark +4 -3
  50. package/grammar/aontu.tmLanguage.json +1 -1
  51. package/package.json +1 -1
  52. package/src/aontu.ts +2 -1
  53. package/src/cli.ts +113 -20
  54. package/src/escape.ts +371 -0
  55. package/src/format.ts +638 -56
  56. package/src/hints.ts +48 -1
  57. package/src/lang.ts +24 -1
  58. package/src/lsp.ts +3 -3
  59. package/src/mcp-server.ts +3 -2
  60. package/src/mod-tool.ts +9 -8
  61. package/src/mod.ts +6 -6
  62. package/src/sigdecl.ts +1 -1
  63. package/src/val/EachFuncVal.ts +1 -3
  64. package/src/val/EmitFuncVal.ts +401 -0
  65. package/src/val/FilterFuncVal.ts +1 -3
  66. package/src/val/FuncBaseVal.ts +18 -0
  67. package/src/val/PackFuncVal.ts +1 -3
  68. package/src/val/PlaceVal.ts +8 -6
  69. package/src/val/StrFuncVal.ts +334 -0
package/dist/format.js CHANGED
@@ -16,10 +16,11 @@ exports.unifiedDiff = unifiedDiff;
16
16
  // trees: a formatter that cannot prove its output is the same document
17
17
  // refuses rather than return it.
18
18
  //
19
- // This is the syntactic tier only (P1): whitespace, commas, quotes,
20
- // bare keys, chains and pair elements, none of which changes the parse
21
- // tree. The lawful tier -- the repeat-the-prefix rewrite that rests on
22
- // the meet -- is P2, and lands behind its own local check.
19
+ // Two tiers. The syntactic (P1): whitespace, commas, quotes, bare
20
+ // keys, chains and pair elements, none of which changes the parse
21
+ // tree. The lawful (P2), over it: repeat the prefix, and merge what
22
+ // repeats -- rewrites that rest on the meet, each checked by the meet
23
+ // in isolation and kept only where the engine agrees.
23
24
  //
24
25
  // The Go twin is go/format.go, function for function; the shared
25
26
  // behaviour is test/spec/fmt.tsv, executed by both spec runners.
@@ -212,26 +213,32 @@ class Reader {
212
213
  opener = false;
213
214
  gap = false;
214
215
  }
216
+ // A blank line before the closer is no paragraph break: nothing
217
+ // follows it, and the layout would drop it anyway.
218
+ while (0 < body.length && 'blank' === body[body.length - 1].t) {
219
+ body.pop();
220
+ }
215
221
  return { body, open };
216
222
  }
217
223
  // One entry: an include, a spread, a pair, or -- as a list element or
218
224
  // at the root -- a value.
219
225
  entry() {
220
226
  const n = this.name(0);
227
+ const at = this.T[this.i].sI;
221
228
  if ('#OD_multisource' === n) {
222
229
  const text = '@' + normStr(this.T[this.i + 1].src);
223
230
  this.i += 2;
224
- return { t: 'include', text };
231
+ return { t: 'include', text, at };
225
232
  }
226
233
  if ('#E&' === n && '#CL' === this.name(1)) {
227
234
  this.i += 2;
228
- return { t: 'spread', value: this.value() };
235
+ return { t: 'spread', value: this.value(), at };
229
236
  }
230
237
  if (this.atKey()) {
231
238
  const tok = this.T[this.i];
232
239
  const opt = '#QM' === this.name(1);
233
240
  this.i += opt ? 3 : 2;
234
- return { t: 'pair', key: keyText(tok), opt, value: this.value() };
241
+ return { t: 'pair', key: keyText(tok), opt, value: this.value(), at };
235
242
  }
236
243
  return this.value();
237
244
  }
@@ -258,12 +265,13 @@ class Reader {
258
265
  if (!this.open(items) && !BINARY[n] && '#LN' !== n && '#CM' !== n) {
259
266
  break;
260
267
  }
268
+ const at = this.T[this.i].sI;
261
269
  if ('#E&' === n && '#CL' === this.name(1)) {
262
270
  if (0 === items.length) {
263
271
  // A chain through a spread, `a: &: integer`. The braces are
264
272
  // the agreed spelling (X-7), so it is read as the map it is.
265
273
  this.i += 2;
266
- return { t: 'map', body: [{ t: 'spread', value: this.value() }] };
274
+ return { t: 'map', body: [{ t: 'spread', value: this.value(), at }], at };
267
275
  }
268
276
  // A sibling spread in a list, `[1 &: 2]`: this value is complete.
269
277
  break;
@@ -283,7 +291,7 @@ class Reader {
283
291
  // operator, or on a line the value continues past. Otherwise
284
292
  // it trails the statement and the caller attaches it.
285
293
  if (this.open(items) || BINARY[this.name(this.significant())]) {
286
- items.push({ t: 'note', text: this.T[this.i].src });
294
+ items.push({ t: 'note', text: this.T[this.i].src, at });
287
295
  this.i++;
288
296
  continue;
289
297
  }
@@ -292,13 +300,13 @@ class Reader {
292
300
  if (BINARY[n]) {
293
301
  items.push({
294
302
  t: 'op', text: this.T[this.i].src,
295
- brk: '#LN' === this.name(-1) || '#LN' === this.name(1),
303
+ brk: '#LN' === this.name(-1) || '#LN' === this.name(1), at,
296
304
  });
297
305
  this.i++;
298
306
  continue;
299
307
  }
300
308
  if (PREFIX[n]) {
301
- items.push({ t: 'prefix', text: this.T[this.i].src });
309
+ items.push({ t: 'prefix', text: this.T[this.i].src, at });
302
310
  this.i++;
303
311
  continue;
304
312
  }
@@ -306,7 +314,7 @@ class Reader {
306
314
  this.i++;
307
315
  const inner = this.seq();
308
316
  this.i++;
309
- items.push({ t: 'paren', inner });
317
+ items.push({ t: 'paren', inner, at });
310
318
  continue;
311
319
  }
312
320
  if ('#TX' === n && '#E(' === this.name(1)) {
@@ -314,25 +322,25 @@ class Reader {
314
322
  this.i += 2;
315
323
  const args = this.seq();
316
324
  this.i++;
317
- items.push({ t: 'call', name, args });
325
+ items.push({ t: 'call', name, args, at });
318
326
  continue;
319
327
  }
320
328
  if ('#OB' === n) {
321
329
  this.i++;
322
330
  const m = this.body('#CB', true);
323
331
  this.i++;
324
- items.push({ t: 'map', body: m.body, open: m.open });
332
+ items.push({ t: 'map', body: m.body, open: m.open, at });
325
333
  continue;
326
334
  }
327
335
  if ('#OS' === n) {
328
336
  this.i++;
329
337
  const l = this.body('#CS', true);
330
338
  this.i++;
331
- items.push({ t: 'list', body: l.body, open: l.open });
339
+ items.push({ t: 'list', body: l.body, open: l.open, at });
332
340
  continue;
333
341
  }
334
342
  if ('#OD_multisource' === n) {
335
- items.push({ t: 'include', text: '@' + normStr(this.T[this.i + 1].src) });
343
+ items.push({ t: 'include', text: '@' + normStr(this.T[this.i + 1].src), at });
336
344
  this.i += 2;
337
345
  continue;
338
346
  }
@@ -348,7 +356,8 @@ class Reader {
348
356
  'note' !== items[0].t) {
349
357
  return items[0];
350
358
  }
351
- return { t: 'expr', items };
359
+ // An empty value, `a:`, is an expression with nothing in it.
360
+ return { t: 'expr', items, at: items[0]?.at };
352
361
  }
353
362
  // Whether the expression so far wants an operand: nothing yet, or an
354
363
  // operator, a prefix or a comment last.
@@ -361,6 +370,7 @@ class Reader {
361
370
  }
362
371
  // The token under the cursor, and the parts glued to it.
363
372
  atom() {
373
+ const at = this.T[this.i].sI;
364
374
  let text = atomText(this.T[this.i]);
365
375
  this.i++;
366
376
  while (GLUE[this.name(0)] &&
@@ -368,7 +378,7 @@ class Reader {
368
378
  text += atomText(this.T[this.i]);
369
379
  this.i++;
370
380
  }
371
- return { t: 'atom', text };
381
+ return { t: 'atom', text, at };
372
382
  }
373
383
  // A call's arguments, or a parenthesis's contents, up to the closing
374
384
  // parenthesis: values separated by commas, with a comment among them
@@ -376,6 +386,7 @@ class Reader {
376
386
  seq() {
377
387
  const out = [];
378
388
  let gap = true;
389
+ let comma = false;
379
390
  for (;;) {
380
391
  const n = this.name(0);
381
392
  if ('' === n || CLOSER[n]) {
@@ -387,9 +398,10 @@ class Reader {
387
398
  }
388
399
  if ('#CA' === n) {
389
400
  if (gap) {
390
- out.push({ t: 'atom', text: 'nil' });
401
+ out.push({ t: 'atom', text: 'nil', sep: comma });
391
402
  }
392
403
  gap = true;
404
+ comma = true;
393
405
  this.i++;
394
406
  continue;
395
407
  }
@@ -398,8 +410,11 @@ class Reader {
398
410
  this.i++;
399
411
  continue;
400
412
  }
401
- out.push(this.value());
413
+ const v = this.value();
414
+ v.sep = comma;
415
+ out.push(v);
402
416
  gap = false;
417
+ comma = false;
403
418
  }
404
419
  return out;
405
420
  }
@@ -508,16 +523,22 @@ function inline(node, tight) {
508
523
  return undefined;
509
524
  }
510
525
  }
526
+ // Arguments on one line, each after the separator the author wrote
527
+ // (§3.6): a comma stays a comma, and a space a space, because the
528
+ // parser reads `must((v) => 0 <= v, "…")` as a run of arguments too.
511
529
  function inlineSeq(items) {
512
- const parts = [];
513
- for (const it of items) {
514
- const s = inline(it, true);
530
+ let out = '';
531
+ for (let k = 0; k < items.length; k++) {
532
+ const s = inline(items[k], true);
515
533
  if (undefined === s) {
516
534
  return undefined;
517
535
  }
518
- parts.push(s);
536
+ out += (0 === k ? '' : sepOf(items[k])) + s;
519
537
  }
520
- return parts.join(', ');
538
+ return out;
539
+ }
540
+ function sepOf(node) {
541
+ return node.sep ? ', ' : ' ';
521
542
  }
522
543
  // Binary operators spaced, prefixes tight (§3.11). An operand is
523
544
  // never directly after an operand: the reader ends a value there.
@@ -570,6 +591,23 @@ class Writer {
570
591
  width() {
571
592
  return width(this.line);
572
593
  }
594
+ // Where the page is, and the lines written since, the current line
595
+ // included: the spelling of one statement, as it stands on the page.
596
+ mark() {
597
+ return this.lines.length;
598
+ }
599
+ since(mark) {
600
+ return this.lines.slice(mark).concat([this.line]).map(rtrim).join('\n') + '\n';
601
+ }
602
+ // The lines since a mark replaced by a text: the spelling before,
603
+ // where a rewrite did not pass its check.
604
+ replace(mark, text) {
605
+ const lines = text.split('\n');
606
+ lines.pop();
607
+ this.line = lines.pop();
608
+ this.lines.length = mark;
609
+ this.lines.push(...lines);
610
+ }
573
611
  finish() {
574
612
  if (!this.started) {
575
613
  return '';
@@ -585,8 +623,11 @@ function rtrim(s) {
585
623
  }
586
624
  // The entries of a body, one per line at the indentation, with the
587
625
  // blank lines the author kept between them (§3.8) -- never at the
588
- // start or the end.
589
- function emitBody(w, body, indent) {
626
+ // start or the end. In STATEMENT position (`stmt`: the root, and the
627
+ // body of a plain map that is itself the value of a statement) a pair
628
+ // is laid out by §3.4, which may repeat its key; anywhere else -- a
629
+ // list, an operand, an argument -- by §3.5 alone.
630
+ function emitBody(w, body, indent, stmt) {
590
631
  let pending = false;
591
632
  let count = 0;
592
633
  for (const node of body) {
@@ -601,6 +642,10 @@ function emitBody(w, body, indent) {
601
642
  w.text(node.text);
602
643
  continue;
603
644
  }
645
+ if (undefined !== stmt && 'pair' === node.t) {
646
+ emitStatement(w, node, indent, stmt, '');
647
+ continue;
648
+ }
604
649
  const e = chain(node);
605
650
  emitValue(w, e, indent);
606
651
  if (undefined !== e.trail) {
@@ -632,10 +677,10 @@ function emitValue(w, node, indent) {
632
677
  emitValue(w, node.value, indent);
633
678
  return;
634
679
  case 'map':
635
- emitBlock(w, '{', '}', node, indent);
680
+ emitBlock(w, '{', '}', node, indent, undefined);
636
681
  return;
637
682
  case 'list':
638
- emitBlock(w, '[', ']', node, indent);
683
+ emitBlock(w, '[', ']', node, indent, undefined);
639
684
  return;
640
685
  case 'expr':
641
686
  emitExpr(w, node.items, indent);
@@ -649,27 +694,36 @@ function emitValue(w, node, indent) {
649
694
  }
650
695
  }
651
696
  // A call, or a parenthesis, that has no one-line form or is too wide
652
- // for the budget. Three shapes. A single container argument hugs the
653
- // parentheses, `close({` ... `})`, and decides its own lines. Arguments
654
- // that each have a one-line form stay on the one line however wide it
655
- // is: the formatter never breaks a line. Otherwise -- an argument that
656
- // is itself several lines, a comment among the arguments -- the
657
- // parenthesis opens a block: one argument per line one level in, the
658
- // closer alone at the opener's level.
697
+ // for the budget. Three shapes. Arguments that are all FLAT -- none
698
+ // holds a container -- stay on the one line however wide it is: a
699
+ // scalar is no narrower on a line of its own, and the formatter never
700
+ // breaks a line. The last argument HUGS the parentheses, `hide({` ...
701
+ // `})`, `close($.E & {` ... `})`, when it is a container, or an
702
+ // expression the author did not break that ends in one, and the
703
+ // arguments before it fit on the opener's line: the container decides
704
+ // its own lines. Otherwise the parenthesis opens a block: one argument
705
+ // per line one level in, the closer alone at the opener's level. A
706
+ // call whose last argument hugs is hugged in turn, `type(close({` ...
707
+ // `}))`: the schema idiom.
659
708
  function emitCall(w, node, indent) {
660
709
  const items = 'call' === node.t ? node.args : node.inner;
661
710
  const open = ('call' === node.t ? node.name : '') + '(';
662
- if (1 === items.length && ('map' === items[0].t || 'list' === items[0].t)) {
663
- w.text(open);
664
- emitValue(w, items[0], indent);
665
- w.text(')');
666
- return;
667
- }
668
711
  const one = inlineSeq(items);
669
- if (undefined !== one) {
712
+ if (undefined !== one && !items.some(holdsContainer)) {
670
713
  w.text(open + one + ')');
671
714
  return;
672
715
  }
716
+ const last = items[items.length - 1];
717
+ if (0 < items.length && hugs(last)) {
718
+ const head = inlineSeq(items.slice(0, -1));
719
+ const lead = '' === head ? '' : head + sepOf(last);
720
+ if (undefined !== head && ('' === head || w.width() + width(open + lead) <= BUDGET)) {
721
+ w.text(open + lead);
722
+ emitValue(w, last, indent);
723
+ w.text(')');
724
+ return;
725
+ }
726
+ }
673
727
  w.text(open);
674
728
  let noted = false;
675
729
  for (let k = 0; k < items.length; k++) {
@@ -690,7 +744,8 @@ function emitCall(w, node, indent) {
690
744
  }
691
745
  w.open(indent + 2, false);
692
746
  emitValue(w, it, indent + 2);
693
- if (items.slice(k + 1).some((x) => 'note' !== x.t)) {
747
+ const next = items.slice(k + 1).find((x) => 'note' !== x.t);
748
+ if (undefined !== next && next.sep) {
694
749
  w.text(',');
695
750
  }
696
751
  noted = false;
@@ -698,10 +753,41 @@ function emitCall(w, node, indent) {
698
753
  w.open(indent, false);
699
754
  w.text(')');
700
755
  }
756
+ // Whether a node holds a container anywhere: the argument has a
757
+ // several-line form of its own.
758
+ function holdsContainer(node) {
759
+ switch (node.t) {
760
+ case 'map':
761
+ case 'list':
762
+ return true;
763
+ case 'call':
764
+ return node.args.some(holdsContainer);
765
+ case 'paren':
766
+ return node.inner.some(holdsContainer);
767
+ case 'expr':
768
+ return node.items.some(holdsContainer);
769
+ default:
770
+ return false;
771
+ }
772
+ }
773
+ // Whether a last argument hugs the parentheses: a container; an
774
+ // expression with no break and no comment whose last operand is one;
775
+ // a call whose own last argument does.
776
+ function hugs(node) {
777
+ if ('map' === node.t || 'list' === node.t) {
778
+ return true;
779
+ }
780
+ if ('call' === node.t) {
781
+ return 0 < node.args.length && hugs(node.args[node.args.length - 1]);
782
+ }
783
+ return 'expr' === node.t &&
784
+ node.items.every((it) => 'note' !== it.t && !('op' === it.t && it.brk)) &&
785
+ hugs(node.items[node.items.length - 1]);
786
+ }
701
787
  // A container on several lines (§3.5): the opener ends its line, the
702
788
  // entries are statements one level in, the closer stands alone. An
703
789
  // empty container is inline whatever the budget says.
704
- function emitBlock(w, open, close, node, indent) {
790
+ function emitBlock(w, open, close, node, indent, stmt) {
705
791
  if (0 === node.body.length && undefined === node.open) {
706
792
  w.text(open + close);
707
793
  return;
@@ -710,7 +796,7 @@ function emitBlock(w, open, close, node, indent) {
710
796
  if (undefined !== node.open) {
711
797
  w.text(' ' + node.open);
712
798
  }
713
- emitBody(w, node.body, indent + 2);
799
+ emitBody(w, node.body, indent + 2, stmt);
714
800
  w.open(indent, false);
715
801
  w.text(close);
716
802
  }
@@ -765,23 +851,420 @@ function emitExpr(w, items, indent) {
765
851
  operand = true;
766
852
  }
767
853
  }
768
- function emit(root) {
854
+ // The entries of a plain map value: a braced map, or a chain, which is
855
+ // a one-entry map. A map with a comment on its opener keeps its braces
856
+ // (§3.7), so it is not plain here; nor is a map holding an include,
857
+ // which the local check cannot follow.
858
+ function plainEntries(v) {
859
+ if ('pair' === v.t) {
860
+ return [v];
861
+ }
862
+ if ('map' !== v.t || undefined !== v.open || v.body.some((e) => 'include' === e.t)) {
863
+ return undefined;
864
+ }
865
+ return v.body;
866
+ }
867
+ // The entries of a statement as they stand once it is merged into a
868
+ // wider map: its trailing comment sunk onto its last entry, so that it
869
+ // travels with the entry it stood beside. Undefined where the value is
870
+ // not a plain map, or the comment has no entry to sit on.
871
+ function members(p) {
872
+ const entries = plainEntries(p.value);
873
+ if (undefined === entries || undefined === p.trail) {
874
+ return entries;
875
+ }
876
+ const last = entries[entries.length - 1];
877
+ if (undefined === last || ('pair' !== last.t && 'spread' !== last.t)) {
878
+ return undefined;
879
+ }
880
+ const trail = undefined === last.trail ? p.trail : last.trail + ' ' + p.trail;
881
+ return entries.slice(0, -1).concat([{ ...last, trail }]);
882
+ }
883
+ // Adjacent statements naming one key, whose values are plain maps, are
884
+ // one map: their entries in order, with the comments and blank lines
885
+ // between the statements travelling with the statement they preceded.
886
+ // Only ADJACENT statements merge -- a `server:` line, something else,
887
+ // then another `server:` line stays as it is, because merging them
888
+ // would move a statement, and the formatter never reorders (§3.13).
889
+ // Nor do two statements merge into a map with two spreads: the engine
890
+ // keeps those as a conjunction, which is not the meet of the two maps.
891
+ // The tree is not changed: a merged statement is a new node that keeps
892
+ // the statements it replaces as its `orig`, its spelling before, and a
893
+ // statement merged somewhere below is copied the same way.
894
+ function mergeRuns(body) {
895
+ const out = [];
896
+ let i = 0;
897
+ while (i < body.length) {
898
+ const first = body[i];
899
+ const entries = 'pair' === first.t ? members(first) : undefined;
900
+ if (undefined === entries) {
901
+ out.push('pair' === first.t ? mergeDeep(first) : first);
902
+ i++;
903
+ continue;
904
+ }
905
+ const group = [first];
906
+ let merged = entries;
907
+ let carry = [];
908
+ let j = i + 1;
909
+ for (; j < body.length; j++) {
910
+ const n = body[j];
911
+ if ('comment' === n.t || 'blank' === n.t) {
912
+ carry.push(n);
913
+ continue;
914
+ }
915
+ const more = 'pair' === n.t && n.key === first.key && n.opt === first.opt
916
+ ? members(n) : undefined;
917
+ if (undefined === more || (spreads(merged) && spreads(more))) {
918
+ break;
919
+ }
920
+ group.push(...carry, n);
921
+ merged = merged.concat(carry, more);
922
+ carry = [];
923
+ }
924
+ if (1 === group.length) {
925
+ out.push(mergeDeep(first));
926
+ i++;
927
+ continue;
928
+ }
929
+ out.push({
930
+ t: 'pair', key: first.key, opt: first.opt,
931
+ value: { t: 'map', body: mergeRuns(merged) }, orig: group,
932
+ });
933
+ i = j - carry.length;
934
+ }
935
+ return out;
936
+ }
937
+ function spreads(entries) {
938
+ return entries.some((e) => 'spread' === e.t);
939
+ }
940
+ // The merge down a statement's plain-map spine: a chain's inner pair,
941
+ // or the entries of a map value, are statements of the map they are
942
+ // in. The statement itself where nothing below it merged.
943
+ function mergeDeep(p) {
944
+ const v = p.value;
945
+ const entries = plainEntries(v);
946
+ if (undefined === entries) {
947
+ return p;
948
+ }
949
+ const body = mergeRuns(entries);
950
+ if (body.length === entries.length && body.every((n, k) => n === entries[k])) {
951
+ return p;
952
+ }
953
+ return { ...p, value: 'pair' === v.t ? body[0] : { ...v, body }, orig: [p] };
954
+ }
955
+ function repeatLines(entries, prefix, indent) {
956
+ if (0 === entries.length || 'comment' === entries[entries.length - 1].t ||
957
+ 1 < entries.filter((e) => 'spread' === e.t).length) {
958
+ return undefined;
959
+ }
960
+ const out = [];
961
+ for (const e of entries) {
962
+ if ('blank' === e.t) {
963
+ out.push({ t: 'blank' });
964
+ continue;
965
+ }
966
+ if ('comment' === e.t) {
967
+ out.push({ t: 'comment', text: e.text });
968
+ continue;
969
+ }
970
+ const trail = undefined === e.trail ? '' : ' ' + e.trail;
971
+ if ('spread' === e.t) {
972
+ // The repeated spread entry is a one-entry map holding only a
973
+ // spread, so by D1's exception it keeps its braces.
974
+ const s = inline(e.value, true);
975
+ if (undefined === s || !fits(indent, prefix + '{ &: ' + s + ' }')) {
976
+ return undefined;
977
+ }
978
+ out.push({ t: 'text', text: prefix + '{ &: ' + s + ' }' + trail });
979
+ continue;
980
+ }
981
+ const head = prefix + pairHead(e, false);
982
+ const s = inline(chain(e.value), false);
983
+ if (undefined !== s && fits(indent, head + s)) {
984
+ out.push({ t: 'text', text: head + s + trail });
985
+ continue;
986
+ }
987
+ const sub = plainEntries(e.value);
988
+ if (undefined === sub) {
989
+ return undefined;
990
+ }
991
+ const lines = repeatLines(sub, head, indent);
992
+ if (undefined === lines) {
993
+ return undefined;
994
+ }
995
+ if ('' !== trail) {
996
+ lines[lines.length - 1].text += trail;
997
+ }
998
+ out.push(...lines);
999
+ }
1000
+ return out;
1001
+ }
1002
+ function fits(indent, text) {
1003
+ return indent + width(text) <= BUDGET;
1004
+ }
1005
+ // A pair in statement position, by §3.4. `prefix` is what stands
1006
+ // before it on its line: the heads of the chain it hangs from, not yet
1007
+ // written. Its value is laid out by §3.5 unless it is a plain map, and
1008
+ // then in this order: a chain, when the map holds exactly one pair
1009
+ // (D1); one line, when that fits the budget; the key repeated over the
1010
+ // entries, when every entry can be one line that way; a braced block
1011
+ // otherwise, whose entries are statements in turn. Whether the
1012
+ // statement was rewritten by this tier -- merged, or repeated -- is
1013
+ // returned, and the outermost such statement is checked: its spelling
1014
+ // on the page against what the syntactic tier writes for the
1015
+ // statements it came from, at the same indentation, which is what
1016
+ // stays on the page when the check fails.
1017
+ function emitStatement(w, p, indent, stmt, prefix) {
1018
+ const mark = w.mark();
1019
+ let rewritten = undefined !== p.orig;
1020
+ const entries = plainEntries(p.value);
1021
+ const head = prefix + pairHead(p, false);
1022
+ const s = undefined === entries ? undefined : inline(p.value, false);
1023
+ if (undefined === entries) {
1024
+ w.text(prefix);
1025
+ emitValue(w, p, indent);
1026
+ }
1027
+ else if (1 === entries.length && 'pair' === entries[0].t) {
1028
+ rewritten = emitStatement(w, entries[0], indent, { meet: stmt.meet, covered: true }, head)
1029
+ || rewritten;
1030
+ }
1031
+ else if (undefined !== s && fits(indent, head + s)) {
1032
+ w.text(head + s);
1033
+ }
1034
+ else {
1035
+ const lines = repeatLines(entries, head, indent);
1036
+ if (undefined !== lines) {
1037
+ let pending = false;
1038
+ let count = 0;
1039
+ for (const line of lines) {
1040
+ if ('blank' === line.t) {
1041
+ pending = 0 < count;
1042
+ continue;
1043
+ }
1044
+ if (0 < count) {
1045
+ w.open(indent, pending);
1046
+ }
1047
+ pending = false;
1048
+ count++;
1049
+ w.text(line.text);
1050
+ }
1051
+ rewritten = true;
1052
+ }
1053
+ else {
1054
+ w.text(head);
1055
+ emitBlock(w, '{', '}', p.value, indent, { meet: stmt.meet, covered: stmt.covered || rewritten });
1056
+ }
1057
+ }
1058
+ if (undefined !== p.trail) {
1059
+ w.text(' ' + p.trail);
1060
+ }
1061
+ if (rewritten && !stmt.covered) {
1062
+ const before = emitAt(p.orig ?? [p], indent);
1063
+ if (!stmt.meet(before, w.since(mark))) {
1064
+ w.replace(mark, before);
1065
+ }
1066
+ }
1067
+ return rewritten;
1068
+ }
1069
+ // The syntactic tier's spelling of some statements at an indentation:
1070
+ // a rewrite's spelling before.
1071
+ function emitAt(nodes, indent) {
1072
+ const w = new Writer();
1073
+ emitBody(w, nodes, indent, undefined);
1074
+ return w.finish();
1075
+ }
1076
+ // The document: by the syntactic tier alone, or with the lawful tier
1077
+ // over it when given its check.
1078
+ function emit(root, meet) {
769
1079
  const w = new Writer();
770
- emitBody(w, root, 0);
1080
+ emitBody(w, undefined === meet ? root : mergeRuns(root), 0, undefined === meet ? undefined : { meet, covered: false });
771
1081
  return w.finish();
772
1082
  }
773
1083
  // ---------------------------------------------------------------------
1084
+ // The lint (§4): what the formatter points at and never touches. Two
1085
+ // rules, both advice: the formatter never renames a key (§4.1) and
1086
+ // never introduces an alias (§4.2), and a rule with a mechanical fix
1087
+ // that keeps the document would belong to §3 instead (§4.3).
1088
+ // The shape width at which a repeat is worth an alias (§4.2): below
1089
+ // it, `{ a:1 }` twice is the shorter spelling. Measured over the use
1090
+ // cases when the lint landed (§7.10).
1091
+ const REPEAT_MIN_WIDTH = 40;
1092
+ function lintOf(root, text) {
1093
+ const out = [];
1094
+ const nodes = root.map(lintNode);
1095
+ for (const n of nodes) {
1096
+ keyCase(n, text, out);
1097
+ }
1098
+ repeats(nodes, text, out);
1099
+ out.sort((a, b) => a.line - b.line || a.col - b.col);
1100
+ return out;
1101
+ }
1102
+ // The tree the lint walks: a chain's inner pair as the one-entry map
1103
+ // it is, so that `a: {b: 1}` and `a: b: 1` -- one document to the
1104
+ // formatter -- are one shape to the lint.
1105
+ function lintNode(node) {
1106
+ if ('pair' === node.t && 'pair' === node.value.t) {
1107
+ return { ...node, value: { t: 'map', body: [node.value], at: node.value.at } };
1108
+ }
1109
+ return node;
1110
+ }
1111
+ function lintChildren(node) {
1112
+ switch (node.t) {
1113
+ case 'pair':
1114
+ case 'spread':
1115
+ return [lintNode(node).value];
1116
+ case 'map':
1117
+ case 'list':
1118
+ return node.body.map(lintNode);
1119
+ case 'call':
1120
+ return node.args;
1121
+ case 'paren':
1122
+ return node.inner;
1123
+ case 'expr':
1124
+ return node.items;
1125
+ default:
1126
+ return [];
1127
+ }
1128
+ }
1129
+ // Line and column, 1-based, of a source index.
1130
+ function lineCol(text, at) {
1131
+ const before = text.slice(0, at);
1132
+ return { line: before.split('\n').length, col: at - before.lastIndexOf('\n') };
1133
+ }
1134
+ // D4 (§4.1): keys are lower-case words, or CamelCase when a key is
1135
+ // several. A bare key holding `_`, or beginning with two capitals, is
1136
+ // reported with the spelling that would follow the form; a quoted key
1137
+ // is a deliberate spelling and a key of underscores alone names
1138
+ // nothing the rule can respell.
1139
+ function keyCase(node, text, out) {
1140
+ if ('pair' === node.t && BARE.test(node.key) && /[A-Za-z]/.test(node.key)) {
1141
+ const why = node.key.includes('_') ? 'holds an underscore'
1142
+ : /^[A-Z][A-Z]/.test(node.key) ? 'begins with capitals' : '';
1143
+ if ('' !== why) {
1144
+ out.push({
1145
+ rule: 'style/key-case', ...lineCol(text, node.at),
1146
+ message: `key ${node.key} ${why}; ${camel(node.key)} would follow the form`,
1147
+ });
1148
+ }
1149
+ }
1150
+ for (const child of lintChildren(node)) {
1151
+ keyCase(child, text, out);
1152
+ }
1153
+ }
1154
+ // The key as lower-case words or CamelCase: `credit_cents` is
1155
+ // `creditCents`, `HTTP_PORT` is `httpPort`, `HTTPServer` is
1156
+ // `httpServer`, `ID` is `id`.
1157
+ function camel(key) {
1158
+ const words = key.split('_').filter((w) => '' !== w)
1159
+ .map((w) => /^[A-Z]+$/.test(w) ? w.toLowerCase() : w);
1160
+ const head = words[0].replace(/^[A-Z]+(?=[A-Z][a-z])/, (run) => run.toLowerCase());
1161
+ return head.charAt(0).toLowerCase() + head.slice(1) +
1162
+ words.slice(1).map((w) => w.charAt(0).toUpperCase() + w.slice(1)).join('');
1163
+ }
1164
+ // D3 (§4.2): a shape written twice can drift, and an alias names it
1165
+ // once. Every map or list whose shape recurs in the file, and whose
1166
+ // shape is REPEAT_MIN_WIDTH or wider, is reported once, at its first
1167
+ // site, with the count and the other sites; the naming is the
1168
+ // author's. A repeat inside a repeat is the outer one's: the walk does
1169
+ // not descend into a shape it reports.
1170
+ function repeats(nodes, text, out) {
1171
+ const counts = new Map();
1172
+ const tally = (node) => {
1173
+ if ('map' === node.t || 'list' === node.t) {
1174
+ const s = shape(node);
1175
+ counts.set(s, (counts.get(s) ?? 0) + 1);
1176
+ }
1177
+ lintChildren(node).forEach(tally);
1178
+ };
1179
+ nodes.forEach(tally);
1180
+ const sites = new Map();
1181
+ const visit = (node) => {
1182
+ if ('map' === node.t || 'list' === node.t) {
1183
+ const s = shape(node);
1184
+ if (2 <= counts.get(s) && REPEAT_MIN_WIDTH <= width(s)) {
1185
+ sites.set(s, (sites.get(s) ?? []).concat([node]));
1186
+ return;
1187
+ }
1188
+ }
1189
+ lintChildren(node).forEach(visit);
1190
+ };
1191
+ nodes.forEach(visit);
1192
+ for (const found of sites.values()) {
1193
+ if (2 <= found.length) {
1194
+ const [first, ...rest] = found.map((n) => lineCol(text, n.at));
1195
+ out.push({
1196
+ rule: 'style/repeat', ...first,
1197
+ message: `this ${found[0].t} is written ${found.length} times (again at ` +
1198
+ rest.map((p) => p.line + ':' + p.col).join(', ') +
1199
+ '); an alias would name it once',
1200
+ });
1201
+ }
1202
+ }
1203
+ }
1204
+ // A node's shape: its spelling with the layout, the comments and, for
1205
+ // a map, the order of its entries taken out, so that two spellings of
1206
+ // one value are one shape, as they are one canon.
1207
+ function shape(node) {
1208
+ switch (node.t) {
1209
+ case 'map':
1210
+ return '{' + node.body.filter(shaped).map((e) => shape(lintNode(e))).sort().join(' ') + '}';
1211
+ case 'list':
1212
+ return '[' + node.body.filter(shaped).map((e) => shape(lintNode(e))).join(' ') + ']';
1213
+ case 'pair':
1214
+ return node.key + (node.opt ? '?' : '') + ':' + shape(node.value);
1215
+ case 'spread':
1216
+ return '&:' + shape(node.value);
1217
+ case 'call':
1218
+ return node.name + '(' + node.args.filter(shaped).map(shape).join(',') + ')';
1219
+ case 'paren':
1220
+ return '(' + node.inner.filter(shaped).map(shape).join(',') + ')';
1221
+ case 'expr':
1222
+ return node.items.filter(shaped).map(shape).join('');
1223
+ default:
1224
+ return node.text;
1225
+ }
1226
+ }
1227
+ function shaped(node) {
1228
+ return 'comment' !== node.t && 'blank' !== node.t && 'note' !== node.t;
1229
+ }
1230
+ // ---------------------------------------------------------------------
774
1231
  // The verb's library surface
775
1232
  function lf(text) {
776
1233
  return text.split('\r\n').join('\n');
777
1234
  }
778
1235
  // The check: the output parses, and to the same tree. Pre-unification
779
- // canon is that tree, positions aside, and every rewrite of this tier
780
- // leaves it unchanged (§7.3).
1236
+ // canon is that tree, positions aside, and every rewrite of the
1237
+ // syntactic tier leaves it unchanged (§7.3).
781
1238
  function sameDocument(root, after) {
782
1239
  const p = parseDoc(after, undefined, undefined);
783
1240
  return undefined === p.errors && root.canon === p.root.canon;
784
1241
  }
1242
+ // The check of a lawful rewrite: the spelling before and the spelling
1243
+ // after, evaluated in isolation, come to the same canon, the same
1244
+ // kinds of failure, and the same outcome of generation (§7.3). Local,
1245
+ // so it needs no include and no capability, and it applies whether or
1246
+ // not the document as a whole evaluates. The kinds, not the count: how
1247
+ // often one unresolved reference is reported depends on the order the
1248
+ // meet took. Generation too, because the engine generates from more
1249
+ // than the canon: a meet of maps with a nil member has refused a key
1250
+ // the same map written once generates.
1251
+ function sameByMeet(before, after) {
1252
+ return meetOf(before) === meetOf(after);
1253
+ }
1254
+ function meetOf(text) {
1255
+ const aontu = engine();
1256
+ const ctx = aontu.ctx({ collect: true });
1257
+ const v = aontu.unify(text, undefined, ctx);
1258
+ const gen = aontu.ctx({ collect: true });
1259
+ const out = aontu.generate(text, undefined, gen);
1260
+ const outcome = undefined !== out ? 'generated'
1261
+ : 0 < ctx.err.length ? kinds(ctx.err) : gen.err[0].why;
1262
+ return v.canon + '\n' + kinds(ctx.err) + '\n' + outcome;
1263
+ }
1264
+ function kinds(errs) {
1265
+ const whys = errs.map((e) => e.why);
1266
+ return whys.filter((x, i) => i === whys.indexOf(x)).sort().join(',');
1267
+ }
785
1268
  function depthFinding() {
786
1269
  return {
787
1270
  code: 'max_depth',
@@ -817,19 +1300,25 @@ function format(src, opts, hooks) {
817
1300
  return { verdict: 'error', errors: parsed.errors };
818
1301
  }
819
1302
  const reader = new Reader(toks);
820
- const root = reader.body('', false).body;
1303
+ const root = unwrap(reader.body('', false).body);
821
1304
  if (reader.deep) {
822
1305
  return { verdict: 'error', errors: [depthFinding()] };
823
1306
  }
824
- const out = emit(unwrap(root));
1307
+ // The syntactic tier first, checked against the parse tree; then the
1308
+ // lawful tier over it, each rewrite checked by the meet.
1309
+ const plain = emit(root, undefined);
825
1310
  const same = hooks?.same ?? sameDocument;
826
- if (!same(parsed.root, out)) {
1311
+ if (!same(parsed.root, plain)) {
827
1312
  return {
828
1313
  verdict: 'error',
829
- errors: [checkFinding(opts?.path, parsed.root.canon, out)],
1314
+ errors: [checkFinding(opts?.path, parsed.root.canon, plain)],
830
1315
  };
831
1316
  }
832
- return { verdict: 'formatted', text: out, changed: out !== src };
1317
+ const out = emit(root, hooks?.meet ?? sameByMeet);
1318
+ return {
1319
+ verdict: 'formatted', text: out, changed: out !== src,
1320
+ findings: opts?.lint ? lintOf(root, text) : [],
1321
+ };
833
1322
  }
834
1323
  // The lines of a text, with a marker on the last when the text does
835
1324
  // not end in a newline: such a line never equals its