aontu 0.55.0 → 0.56.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. package/README.md +3 -3
  2. package/dist/agentsmd.d.ts +1 -0
  3. package/dist/agentsmd.js +10 -3
  4. package/dist/agentsmd.js.map +1 -1
  5. package/dist/aontu.d.ts +3 -2
  6. package/dist/aontu.js +5 -2
  7. package/dist/aontu.js.map +1 -1
  8. package/dist/cli.d.ts +5 -2
  9. package/dist/cli.js +236 -49
  10. package/dist/cli.js.map +1 -1
  11. package/dist/diff.d.ts +1 -0
  12. package/dist/diff.js +3 -2
  13. package/dist/diff.js.map +1 -1
  14. package/dist/format.d.ts +17 -0
  15. package/dist/format.js +1000 -0
  16. package/dist/format.js.map +1 -0
  17. package/dist/hints.js +4 -0
  18. package/dist/hints.js.map +1 -1
  19. package/dist/jsonschema.d.ts +1 -0
  20. package/dist/jsonschema.js +3 -2
  21. package/dist/jsonschema.js.map +1 -1
  22. package/dist/lang.js +64 -5
  23. package/dist/lang.js.map +1 -1
  24. package/dist/lsp.d.ts +2 -1
  25. package/dist/lsp.js +1 -1
  26. package/dist/lsp.js.map +1 -1
  27. package/dist/mcp.js +6 -4
  28. package/dist/mcp.js.map +1 -1
  29. package/dist/patch.d.ts +1 -0
  30. package/dist/patch.js +1 -0
  31. package/dist/patch.js.map +1 -1
  32. package/dist/query.d.ts +1 -0
  33. package/dist/query.js +4 -3
  34. package/dist/query.js.map +1 -1
  35. package/dist/reach.d.ts +1 -0
  36. package/dist/reach.js +3 -2
  37. package/dist/reach.js.map +1 -1
  38. package/dist/relation.d.ts +1 -0
  39. package/dist/relation.js +3 -2
  40. package/dist/relation.js.map +1 -1
  41. package/dist/std.js +2 -1
  42. package/dist/std.js.map +1 -1
  43. package/dist/subsume.d.ts +1 -0
  44. package/dist/subsume.js +3 -2
  45. package/dist/subsume.js.map +1 -1
  46. package/dist/trim.d.ts +1 -0
  47. package/dist/trim.js +3 -2
  48. package/dist/trim.js.map +1 -1
  49. package/dist/tsconfig.tsbuildinfo +1 -1
  50. package/dist/type.d.ts +1 -0
  51. package/dist/type.js.map +1 -1
  52. package/dist/utility.d.ts +8 -2
  53. package/dist/utility.js +9 -1
  54. package/dist/utility.js.map +1 -1
  55. package/dist/vet.d.ts +1 -0
  56. package/dist/vet.js +1 -1
  57. package/dist/vet.js.map +1 -1
  58. package/dist/view.d.ts +2 -1
  59. package/dist/view.js +379 -8
  60. package/dist/view.js.map +1 -1
  61. package/grammar/aontu.abnf +155 -0
  62. package/package.json +2 -1
  63. package/skill/grammar-card.md +5 -2
  64. package/src/agentsmd.ts +15 -3
  65. package/src/aontu.ts +7 -1
  66. package/src/cli.ts +267 -53
  67. package/src/diff.ts +7 -2
  68. package/src/format.ts +1146 -0
  69. package/src/hints.ts +5 -0
  70. package/src/jsonschema.ts +7 -1
  71. package/src/lang.ts +69 -5
  72. package/src/lsp.ts +5 -3
  73. package/src/mcp.ts +6 -4
  74. package/src/patch.ts +7 -0
  75. package/src/query.ts +8 -4
  76. package/src/reach.ts +7 -2
  77. package/src/relation.ts +7 -2
  78. package/src/std.ts +2 -1
  79. package/src/subsume.ts +7 -2
  80. package/src/trim.ts +7 -2
  81. package/src/type.ts +8 -0
  82. package/src/utility.ts +27 -2
  83. package/src/vet.ts +9 -3
  84. package/src/view.ts +442 -9
package/src/format.ts ADDED
@@ -0,0 +1,1146 @@
1
+ /* Copyright (c) 2026 Richard Rodger, MIT License */
2
+
3
+ // THE SOURCE FORMATTER (docs/design/FMT.0.md): `aontu fmt`, in the
4
+ // tradition of gofmt. One agreed form for Aontu source, so that layout
5
+ // is never argued about and a diff shows only what changed.
6
+ //
7
+ // It reads the token stream the parser reads -- the lex subscriber the
8
+ // parser stack exposes -- so it sees what the value tree throws away:
9
+ // comments, blank lines, the quote a string used, the spelling of a
10
+ // number. From that stream it builds a layout tree, decides the shape
11
+ // of every container by the rules of the note's §3, and emits. Before
12
+ // returning it re-parses what it wrote and compares the two parse
13
+ // trees: a formatter that cannot prove its output is the same document
14
+ // refuses rather than return it.
15
+ //
16
+ // This is the syntactic tier only (P1): whitespace, commas, quotes,
17
+ // bare keys, chains and pair elements, none of which changes the parse
18
+ // tree. The lawful tier -- the repeat-the-prefix rewrite that rests on
19
+ // the meet -- is P2, and lands behind its own local check.
20
+ //
21
+ // The Go twin is go/format.go, function for function; the shared
22
+ // behaviour is test/spec/fmt.tsv, executed by both spec runners.
23
+
24
+ import { Aontu } from './aontu'
25
+ import { failureFinding } from './vet'
26
+ import type { VetFinding } from './vet'
27
+ import type { Resolver } from './type'
28
+
29
+
30
+ // The packing budget (§3.1). It decides which of two legal spellings
31
+ // to use, one line or several, and nothing else: the formatter never
32
+ // breaks a line, so a value wider than this stays as wide as it is.
33
+ const BUDGET = 80
34
+
35
+ // THE DEPTH BUDGET. The layout is recursive, as the tree it reads is,
36
+ // and the canonical port's stack is finite: past the evaluation budget
37
+ // of 1000 levels -- the depth at which unification itself refuses --
38
+ // the formatter stops reading and refuses, so a pathological document
39
+ // is a finding rather than a crash.
40
+ const MAX_DEPTH = 1000
41
+
42
+ export type FormatOptions = {
43
+ // The file's name, for the site of a parse failure.
44
+ path?: string
45
+ }
46
+
47
+ // The self-check, injectable so the refusal it guards can be exercised
48
+ // (the `hooks` precedent of `view`): a formatter that is right never
49
+ // takes that arm on its own.
50
+ export type FormatHooks = {
51
+ same?: (root: any, after: string) => boolean
52
+ }
53
+
54
+ export type FormatReport =
55
+ | { verdict: 'formatted', text: string, changed: boolean }
56
+ | { verdict: 'error', errors: VetFinding[] }
57
+
58
+
59
+ // ---------------------------------------------------------------------
60
+ // The tokens
61
+
62
+ type Tok = { name: string, src: string, val: any, sI: number }
63
+
64
+ // EVERY INCLUDE RESOLVES TO NOTHING. The formatter reads the file it is
65
+ // given and no other (§3.13), so `@"..."` is answered from memory with
66
+ // an empty source: the directive parses, the include is a token like
67
+ // any other, and no capability is needed because no file is read.
68
+ const stubResolver: Resolver = ((spec: any) => ({
69
+ ...spec, kind: 'aon', full: '__fmt__.aon', src: '', found: true, search: [],
70
+ })) as any
71
+
72
+ // ONE ENGINE, ONE SUBSCRIBER. The parser's subscriber list is
73
+ // append-only, so the subscription is made once and writes to
74
+ // whichever sink the current parse installed; the sink is cleared
75
+ // before the parse returns, so the check's re-parse collects nothing.
76
+ let ENGINE: Aontu | undefined
77
+ let SINK: Tok[] | undefined
78
+
79
+ function engine(): Aontu {
80
+ if (undefined === ENGINE) {
81
+ ENGINE = new Aontu({ resolver: stubResolver })
82
+ ENGINE.lang.jsonic.sub({
83
+ lex: (tkn: any) => {
84
+ // Spaces carry nothing the layout needs, and the end token
85
+ // arrives once per nested parse -- the stub's empty includes
86
+ // among them -- so both are dropped here rather than skipped
87
+ // everywhere below.
88
+ if (undefined !== SINK && '#SP' !== tkn.name && '#ZZ' !== tkn.name) {
89
+ SINK.push({ name: tkn.name, src: tkn.src, val: tkn.val, sI: tkn.sI })
90
+ }
91
+ },
92
+ })
93
+ }
94
+ return ENGINE
95
+ }
96
+
97
+
98
+ type Parsed = { root?: any, errors?: VetFinding[] }
99
+
100
+ // One parse, with the token stream collected when a sink is given. The
101
+ // failure shape is the one every verb reports (`view`'s load).
102
+ function parseDoc(src: string, path: string | undefined, sink: Tok[] | undefined): Parsed {
103
+ const aontu = engine()
104
+ const ctx = aontu.ctx({ collect: true })
105
+ SINK = sink
106
+ let parsed: any
107
+ try {
108
+ parsed = aontu.parse(src, undefined === path ? undefined : { path }, ctx)
109
+ }
110
+ finally {
111
+ SINK = undefined
112
+ }
113
+ if (0 < ctx.err.length) {
114
+ return { errors: [failureFinding(ctx, path, parsed)] }
115
+ }
116
+ return { root: parsed }
117
+ }
118
+
119
+
120
+ // ---------------------------------------------------------------------
121
+ // The layout tree
122
+
123
+ // One node shape for the whole tree, so the Go twin is one struct:
124
+ // the kind says which fields are meaningful.
125
+ type Node = {
126
+ t: 'pair' | 'spread' | 'include' | 'comment' | 'blank' | 'map' | 'list'
127
+ | 'atom' | 'call' | 'paren' | 'expr' | 'op' | 'prefix' | 'note'
128
+
129
+ // atom, include, comment, note, op, prefix: the text as written,
130
+ // normalised where §3.9 says (quotes), and nothing else.
131
+ text?: string
132
+
133
+ // pair: the key as it will be written, the optional marker, and the
134
+ // value; spread: the value.
135
+ key?: string
136
+ opt?: boolean
137
+ value?: Node
138
+
139
+ // map, list: the entries, and the comment on the opener's line.
140
+ body?: Node[]
141
+ open?: string
142
+
143
+ // call: the name and the arguments; paren: what it groups, which the
144
+ // parser reads as a call's argument list does (commas and all).
145
+ name?: string
146
+ args?: Node[]
147
+ inner?: Node[]
148
+
149
+ // expr: operands, binary operators, prefix operators and notes (a
150
+ // comment inside the expression), in source order.
151
+ items?: Node[]
152
+
153
+ // op: the author broke the line at this operator (§3.11).
154
+ brk?: boolean
155
+
156
+ // A comment on the last line of this entry.
157
+ trail?: string
158
+ }
159
+
160
+ const BINARY: Record<string, boolean> = { '#E&': true, '#E|': true, '#E+': true }
161
+ const PREFIX: Record<string, boolean> = { '#E*': true, '#E-': true }
162
+ const KEYISH: Record<string, boolean> = { '#TX': true, '#ST': true, '#NR': true, '#VL': true }
163
+ const CLOSER: Record<string, boolean> = { '#CB': true, '#CS': true, '#E)': true }
164
+
165
+ // The parts of one atom: a reference is `$`, dots and segments lexed
166
+ // one by one, and a bare word with a dot in it is the same run; what
167
+ // was adjacent in the source stays glued.
168
+ const GLUE: Record<string, boolean> = {
169
+ '#TX': true, '#ST': true, '#NR': true, '#VL': true, '#E.': true, '#E$': true,
170
+ }
171
+
172
+ const BARE = /^[A-Za-z_][A-Za-z0-9_]*$/
173
+
174
+ // A single-quoted string becomes double-quoted unless it holds a double
175
+ // quote, which the swap would have to escape (§3.9). The body is copied
176
+ // as written: the escapes are the same under both quotes.
177
+ function normStr(src: string): string {
178
+ if ("'" === src[0]) {
179
+ const body = src.slice(1, -1)
180
+ return body.includes('"') ? src : '"' + body + '"'
181
+ }
182
+ return src
183
+ }
184
+
185
+ function atomText(tok: Tok): string {
186
+ return '#ST' === tok.name ? normStr(tok.src) : tok.src
187
+ }
188
+
189
+ // A quoted key whose text is a legal bare key is written bare; the
190
+ // keywords are legal keys too (`string: 1` is the key `string`), so no
191
+ // word is reserved. Anything else keeps its spelling.
192
+ function keyText(tok: Tok): string {
193
+ if ('#ST' === tok.name) {
194
+ return BARE.test(tok.val) ? tok.val : normStr(tok.src)
195
+ }
196
+ return tok.src
197
+ }
198
+
199
+ function newlines(src: string): number {
200
+ return src.split('\n').length - 1
201
+ }
202
+
203
+
204
+ class Reader {
205
+ T: Tok[]
206
+ i = 0
207
+ depth = 0
208
+ // Past the depth budget: the reader answers '' for every token from
209
+ // here on, so every loop unwinds, and the document is refused.
210
+ deep = false
211
+
212
+ constructor(toks: Tok[]) {
213
+ this.T = toks
214
+ }
215
+
216
+ // The name of the token k ahead, or '' past the end.
217
+ name(k: number): string {
218
+ const t = this.T[this.i + k]
219
+ return this.deep || undefined === t ? '' : t.name
220
+ }
221
+
222
+ // The offset of the next token that is not a line run or a comment.
223
+ significant(): number {
224
+ let k = 0
225
+ while ('#LN' === this.name(k) || '#CM' === this.name(k)) {
226
+ k++
227
+ }
228
+ return k
229
+ }
230
+
231
+ // A key followed by a colon, the optional marker allowed between.
232
+ atKey(): boolean {
233
+ return KEYISH[this.name(0)] && ('#CL' === this.name(1) ||
234
+ ('#QM' === this.name(1) && '#CL' === this.name(2)))
235
+ }
236
+
237
+ // The entries of a container up to its closer, or of the document up
238
+ // to its end. Comments attach by the rules of §3.7: on the line of
239
+ // the entry that precedes them, or of the opener, they trail it;
240
+ // alone on a line they stand as entries and precede what follows.
241
+ body(close: string, opened: boolean): { body: Node[], open?: string } {
242
+ const body: Node[] = []
243
+ let open: string | undefined
244
+ let last: Node | undefined
245
+ let opener = opened
246
+ // Nothing since the opener or the last comma: a comma here is an
247
+ // empty element, which the parser reads as nil in a list.
248
+ let gap = true
249
+ for (;;) {
250
+ const n = this.name(0)
251
+ // The closer, or the end: the parser accepts a container the
252
+ // source never closed (`a: {` is `{"a":{}}`).
253
+ if ('' === n || n === close) {
254
+ break
255
+ }
256
+ if ('#LN' === n) {
257
+ if (1 < newlines(this.T[this.i].src) && 0 < body.length &&
258
+ 'blank' !== body[body.length - 1].t) {
259
+ body.push({ t: 'blank' })
260
+ }
261
+ last = undefined
262
+ opener = false
263
+ this.i++
264
+ continue
265
+ }
266
+ if ('#CA' === n) {
267
+ if (gap && '#CS' === close) {
268
+ const nil: Node = { t: 'atom', text: 'nil' }
269
+ body.push(nil)
270
+ last = nil
271
+ }
272
+ gap = true
273
+ this.i++
274
+ continue
275
+ }
276
+ if ('#CM' === n) {
277
+ const text = this.T[this.i].src
278
+ if (undefined !== last) {
279
+ last.trail = text
280
+ }
281
+ else if (opener) {
282
+ open = text
283
+ }
284
+ else {
285
+ body.push({ t: 'comment', text })
286
+ }
287
+ this.i++
288
+ continue
289
+ }
290
+ if (CLOSER[n]) {
291
+ // A closer that is not this container's: the parser ignores a
292
+ // stray one at the root (`a: 1 }` is `{"a":1}`), and so does
293
+ // this.
294
+ this.i++
295
+ continue
296
+ }
297
+ const e = this.entry()
298
+ body.push(e)
299
+ last = e
300
+ opener = false
301
+ gap = false
302
+ }
303
+ return { body, open }
304
+ }
305
+
306
+ // One entry: an include, a spread, a pair, or -- as a list element or
307
+ // at the root -- a value.
308
+ entry(): Node {
309
+ const n = this.name(0)
310
+ if ('#OD_multisource' === n) {
311
+ const text = '@' + normStr(this.T[this.i + 1].src)
312
+ this.i += 2
313
+ return { t: 'include', text }
314
+ }
315
+ if ('#E&' === n && '#CL' === this.name(1)) {
316
+ this.i += 2
317
+ return { t: 'spread', value: this.value() }
318
+ }
319
+ if (this.atKey()) {
320
+ const tok = this.T[this.i]
321
+ const opt = '#QM' === this.name(1)
322
+ this.i += opt ? 3 : 2
323
+ return { t: 'pair', key: keyText(tok), opt, value: this.value() }
324
+ }
325
+ return this.value()
326
+ }
327
+
328
+ // A value: operands and operators up to whatever ends it -- a
329
+ // separator, a closer, the end, or a line run that no operator
330
+ // continues past.
331
+ value(): Node {
332
+ if (MAX_DEPTH < ++this.depth) {
333
+ this.deep = true
334
+ }
335
+ const v = this.valueAt()
336
+ this.depth--
337
+ return v
338
+ }
339
+
340
+ valueAt(): Node {
341
+ const items: Node[] = []
342
+ for (;;) {
343
+ const n = this.name(0)
344
+ if ('' === n || '#CA' === n || CLOSER[n]) {
345
+ break
346
+ }
347
+ // An operand directly after an operand is the next element of a
348
+ // list, `[1 -2]`, `[{a:1} {b:2}]`: this value is complete.
349
+ if (!this.open(items) && !BINARY[n] && '#LN' !== n && '#CM' !== n) {
350
+ break
351
+ }
352
+ if ('#E&' === n && '#CL' === this.name(1)) {
353
+ if (0 === items.length) {
354
+ // A chain through a spread, `a: &: integer`. The braces are
355
+ // the agreed spelling (X-7), so it is read as the map it is.
356
+ this.i += 2
357
+ return { t: 'map', body: [{ t: 'spread', value: this.value() }] }
358
+ }
359
+ // A sibling spread in a list, `[1 &: 2]`: this value is complete.
360
+ break
361
+ }
362
+ if ('#LN' === n) {
363
+ // A break the author put before the value, after an operator
364
+ // (`a: 1 &\n 2`) or before one (`a: 1\n | 2`), or after a
365
+ // comment inside the value; anything else ends the value.
366
+ if (this.open(items) || BINARY[this.name(this.significant())]) {
367
+ this.i++
368
+ continue
369
+ }
370
+ break
371
+ }
372
+ if ('#CM' === n) {
373
+ // A comment inside the value: after the colon, after an
374
+ // operator, or on a line the value continues past. Otherwise
375
+ // it trails the statement and the caller attaches it.
376
+ if (this.open(items) || BINARY[this.name(this.significant())]) {
377
+ items.push({ t: 'note', text: this.T[this.i].src })
378
+ this.i++
379
+ continue
380
+ }
381
+ break
382
+ }
383
+ if (BINARY[n]) {
384
+ items.push({
385
+ t: 'op', text: this.T[this.i].src,
386
+ brk: '#LN' === this.name(-1) || '#LN' === this.name(1),
387
+ })
388
+ this.i++
389
+ continue
390
+ }
391
+ if (PREFIX[n]) {
392
+ items.push({ t: 'prefix', text: this.T[this.i].src })
393
+ this.i++
394
+ continue
395
+ }
396
+ if ('#E(' === n) {
397
+ this.i++
398
+ const inner = this.seq()
399
+ this.i++
400
+ items.push({ t: 'paren', inner })
401
+ continue
402
+ }
403
+ if ('#TX' === n && '#E(' === this.name(1)) {
404
+ const name = this.T[this.i].src
405
+ this.i += 2
406
+ const args = this.seq()
407
+ this.i++
408
+ items.push({ t: 'call', name, args })
409
+ continue
410
+ }
411
+ if ('#OB' === n) {
412
+ this.i++
413
+ const m = this.body('#CB', true)
414
+ this.i++
415
+ items.push({ t: 'map', body: m.body, open: m.open })
416
+ continue
417
+ }
418
+ if ('#OS' === n) {
419
+ this.i++
420
+ const l = this.body('#CS', true)
421
+ this.i++
422
+ items.push({ t: 'list', body: l.body, open: l.open })
423
+ continue
424
+ }
425
+ if ('#OD_multisource' === n) {
426
+ items.push({ t: 'include', text: '@' + normStr(this.T[this.i + 1].src) })
427
+ this.i += 2
428
+ continue
429
+ }
430
+ if (this.atKey()) {
431
+ // A pair in value position is a chain, `a: b: 1`, and it is
432
+ // the whole of the value.
433
+ items.push(this.entry())
434
+ break
435
+ }
436
+ items.push(this.atom())
437
+ }
438
+ if (1 === items.length && 'op' !== items[0].t && 'prefix' !== items[0].t &&
439
+ 'note' !== items[0].t) {
440
+ return items[0]
441
+ }
442
+ return { t: 'expr', items }
443
+ }
444
+
445
+ // Whether the expression so far wants an operand: nothing yet, or an
446
+ // operator, a prefix or a comment last.
447
+ open(items: Node[]): boolean {
448
+ if (0 === items.length) {
449
+ return true
450
+ }
451
+ const t = items[items.length - 1].t
452
+ return 'op' === t || 'prefix' === t || 'note' === t
453
+ }
454
+
455
+ // The token under the cursor, and the parts glued to it.
456
+ atom(): Node {
457
+ let text = atomText(this.T[this.i])
458
+ this.i++
459
+ while (GLUE[this.name(0)] &&
460
+ this.T[this.i - 1].sI + this.T[this.i - 1].src.length === this.T[this.i].sI) {
461
+ text += atomText(this.T[this.i])
462
+ this.i++
463
+ }
464
+ return { t: 'atom', text }
465
+ }
466
+
467
+ // A call's arguments, or a parenthesis's contents, up to the closing
468
+ // parenthesis: values separated by commas, with a comment among them
469
+ // kept as a note.
470
+ seq(): Node[] {
471
+ const out: Node[] = []
472
+ let gap = true
473
+ for (;;) {
474
+ const n = this.name(0)
475
+ if ('' === n || CLOSER[n]) {
476
+ break
477
+ }
478
+ if ('#LN' === n) {
479
+ this.i++
480
+ continue
481
+ }
482
+ if ('#CA' === n) {
483
+ if (gap) {
484
+ out.push({ t: 'atom', text: 'nil' })
485
+ }
486
+ gap = true
487
+ this.i++
488
+ continue
489
+ }
490
+ if ('#CM' === n) {
491
+ out.push({ t: 'note', text: this.T[this.i].src })
492
+ this.i++
493
+ continue
494
+ }
495
+ out.push(this.value())
496
+ gap = false
497
+ }
498
+ return out
499
+ }
500
+ }
501
+
502
+
503
+ // THE ROOT MAP HAS NO BRACES (§3.12). A document written as one braced
504
+ // map is its entries; the comments on the braces' lines become entries
505
+ // of their own, where nothing is lost.
506
+ function unwrap(root: Node[]): Node[] {
507
+ const entries = root.filter((n) => 'comment' !== n.t && 'blank' !== n.t)
508
+ if (1 !== entries.length || 'map' !== entries[0].t) {
509
+ return root
510
+ }
511
+ const m = entries[0]
512
+ const out: Node[] = []
513
+ for (const n of root) {
514
+ if (n !== m) {
515
+ out.push(n)
516
+ continue
517
+ }
518
+ if (undefined !== m.open) {
519
+ out.push({ t: 'comment', text: m.open })
520
+ }
521
+ out.push(...m.body!)
522
+ if (undefined !== m.trail) {
523
+ out.push({ t: 'comment', text: m.trail })
524
+ }
525
+ }
526
+ return out
527
+ }
528
+
529
+
530
+ // ---------------------------------------------------------------------
531
+ // The layout
532
+
533
+ // D1: a one-pair map in value position is written as a chain, and a
534
+ // one-pair map as a list element as a pair element. A map whose only
535
+ // entry is a spread keeps its braces (X-7), and one holding a comment
536
+ // keeps them too, because the comment needs the lines. A trailing
537
+ // comment on the map's line joins the pair's own.
538
+ function chain(node: Node): Node {
539
+ if ('map' !== node.t || undefined !== node.open || 1 !== node.body!.length ||
540
+ 'pair' !== node.body![0].t) {
541
+ return node
542
+ }
543
+ const p = node.body![0]
544
+ if (undefined === node.trail) {
545
+ return p
546
+ }
547
+ return { ...p, trail: undefined === p.trail ? node.trail : p.trail + ' ' + node.trail }
548
+ }
549
+
550
+ function width(s: string): number {
551
+ return Array.from(s).length
552
+ }
553
+
554
+ function pairHead(node: Node, tight: boolean): string {
555
+ return node.key! + (node.opt ? '?' : '') + (tight ? ':' : ': ')
556
+ }
557
+
558
+ // The one-line spelling of a node, or undefined where it has none: a
559
+ // comment, a blank line, a break the author kept, a string that spans
560
+ // lines. `tight` is the inline form of a pair, `a:1`, used inside a
561
+ // container; a statement's pair is `a: 1`.
562
+ function inline(node: Node, tight: boolean): string | undefined {
563
+ if (undefined !== node.trail) {
564
+ return undefined
565
+ }
566
+ switch (node.t) {
567
+ case 'atom':
568
+ case 'include':
569
+ return node.text!.includes('\n') ? undefined : node.text
570
+ case 'pair': {
571
+ const v = inline(chain(node.value!), tight)
572
+ return undefined === v ? undefined : pairHead(node, tight) + v
573
+ }
574
+ case 'spread': {
575
+ // `{ &: integer }`, padded inside braces too: the marker reads as
576
+ // a marker and not as a key.
577
+ const v = inline(node.value!, tight)
578
+ return undefined === v ? undefined : '&: ' + v
579
+ }
580
+ case 'map':
581
+ case 'list': {
582
+ if (undefined !== node.open) {
583
+ return undefined
584
+ }
585
+ const parts: string[] = []
586
+ for (const e of node.body!) {
587
+ const s = inline('list' === node.t ? chain(e) : e, true)
588
+ if (undefined === s) {
589
+ return undefined
590
+ }
591
+ parts.push(s)
592
+ }
593
+ if ('list' === node.t) {
594
+ return '[' + parts.join(' ') + ']'
595
+ }
596
+ return 0 === parts.length ? '{}' : '{ ' + parts.join(' ') + ' }'
597
+ }
598
+ case 'call': {
599
+ const a = inlineSeq(node.args!)
600
+ return undefined === a ? undefined : node.name + '(' + a + ')'
601
+ }
602
+ case 'paren': {
603
+ const a = inlineSeq(node.inner!)
604
+ return undefined === a ? undefined : '(' + a + ')'
605
+ }
606
+ case 'expr':
607
+ return inlineExpr(node.items!)
608
+ default:
609
+ // comment, blank: never on a line with anything else.
610
+ return undefined
611
+ }
612
+ }
613
+
614
+ function inlineSeq(items: Node[]): string | undefined {
615
+ const parts: string[] = []
616
+ for (const it of items) {
617
+ const s = inline(it, true)
618
+ if (undefined === s) {
619
+ return undefined
620
+ }
621
+ parts.push(s)
622
+ }
623
+ return parts.join(', ')
624
+ }
625
+
626
+ // Binary operators spaced, prefixes tight (§3.11). An operand is
627
+ // never directly after an operand: the reader ends a value there.
628
+ function inlineExpr(items: Node[]): string | undefined {
629
+ let out = ''
630
+ for (const it of items) {
631
+ if ('note' === it.t || ('op' === it.t && it.brk)) {
632
+ return undefined
633
+ }
634
+ if ('op' === it.t) {
635
+ out += ' ' + it.text + ' '
636
+ continue
637
+ }
638
+ if ('prefix' === it.t) {
639
+ out += it.text
640
+ continue
641
+ }
642
+ const s = inline(it, true)
643
+ if (undefined === s) {
644
+ return undefined
645
+ }
646
+ out += s
647
+ }
648
+ return out
649
+ }
650
+
651
+
652
+ class Writer {
653
+ lines: string[] = []
654
+ line = ''
655
+ started = false
656
+
657
+ // A new line at an indentation, after a blank one when asked.
658
+ open(indent: number, blank: boolean): void {
659
+ if (this.started) {
660
+ this.lines.push(rtrim(this.line))
661
+ if (blank) {
662
+ this.lines.push('')
663
+ }
664
+ }
665
+ this.line = ' '.repeat(indent)
666
+ this.started = true
667
+ }
668
+
669
+ text(s: string): void {
670
+ this.line += s
671
+ }
672
+
673
+ // Nothing on the line yet but its indentation.
674
+ fresh(): boolean {
675
+ return '' === this.line.trim()
676
+ }
677
+
678
+ width(): number {
679
+ return width(this.line)
680
+ }
681
+
682
+ finish(): string {
683
+ if (!this.started) {
684
+ return ''
685
+ }
686
+ this.lines.push(rtrim(this.line))
687
+ return this.lines.join('\n') + '\n'
688
+ }
689
+ }
690
+
691
+ // A line never ends in a space: an operator the author left dangling
692
+ // (`a: 1 &`, which the parser accepts) would otherwise leave one.
693
+ function rtrim(s: string): string {
694
+ return s.replace(/ +$/, '')
695
+ }
696
+
697
+
698
+ // The entries of a body, one per line at the indentation, with the
699
+ // blank lines the author kept between them (§3.8) -- never at the
700
+ // start or the end.
701
+ function emitBody(w: Writer, body: Node[], indent: number): void {
702
+ let pending = false
703
+ let count = 0
704
+ for (const node of body) {
705
+ if ('blank' === node.t) {
706
+ pending = 0 < count
707
+ continue
708
+ }
709
+ w.open(indent, pending)
710
+ pending = false
711
+ count++
712
+ if ('comment' === node.t) {
713
+ w.text(node.text!)
714
+ continue
715
+ }
716
+ const e = chain(node)
717
+ emitValue(w, e, indent)
718
+ if (undefined !== e.trail) {
719
+ w.text(' ' + e.trail)
720
+ }
721
+ }
722
+ }
723
+
724
+ // A value onto the current line: its one-line spelling when there is
725
+ // one and it fits the budget, and otherwise its several-line form,
726
+ // which for a scalar is the same text, too wide and unbreakable.
727
+ function emitValue(w: Writer, node: Node, indent: number): void {
728
+ const s = inline(node, false)
729
+ if (undefined !== s && w.width() + width(s) <= BUDGET) {
730
+ w.text(s)
731
+ return
732
+ }
733
+ switch (node.t) {
734
+ case 'pair': {
735
+ w.text(pairHead(node, false))
736
+ const v = chain(node.value!)
737
+ emitValue(w, v, indent)
738
+ if (undefined !== v.trail) {
739
+ w.text(' ' + v.trail)
740
+ }
741
+ return
742
+ }
743
+ case 'spread':
744
+ w.text('&: ')
745
+ emitValue(w, node.value!, indent)
746
+ return
747
+ case 'map':
748
+ emitBlock(w, '{', '}', node, indent)
749
+ return
750
+ case 'list':
751
+ emitBlock(w, '[', ']', node, indent)
752
+ return
753
+ case 'expr':
754
+ emitExpr(w, node.items!, indent)
755
+ return
756
+ case 'call':
757
+ case 'paren':
758
+ emitCall(w, node, indent)
759
+ return
760
+ default:
761
+ w.text(node.text!)
762
+ }
763
+ }
764
+
765
+ // A call, or a parenthesis, that has no one-line form or is too wide
766
+ // for the budget. Three shapes. A single container argument hugs the
767
+ // parentheses, `close({` ... `})`, and decides its own lines. Arguments
768
+ // that each have a one-line form stay on the one line however wide it
769
+ // is: the formatter never breaks a line. Otherwise -- an argument that
770
+ // is itself several lines, a comment among the arguments -- the
771
+ // parenthesis opens a block: one argument per line one level in, the
772
+ // closer alone at the opener's level.
773
+ function emitCall(w: Writer, node: Node, indent: number): void {
774
+ const items = 'call' === node.t ? node.args! : node.inner!
775
+ const open = ('call' === node.t ? node.name! : '') + '('
776
+ if (1 === items.length && ('map' === items[0].t || 'list' === items[0].t)) {
777
+ w.text(open)
778
+ emitValue(w, items[0], indent)
779
+ w.text(')')
780
+ return
781
+ }
782
+ const one = inlineSeq(items)
783
+ if (undefined !== one) {
784
+ w.text(open + one + ')')
785
+ return
786
+ }
787
+ w.text(open)
788
+ let noted = false
789
+ for (let k = 0; k < items.length; k++) {
790
+ const it = items[k]
791
+ if ('note' === it.t) {
792
+ // A comment among the arguments trails the line it was on -- the
793
+ // opener's, or an argument's -- and one that followed another
794
+ // comment keeps its own line.
795
+ if (noted) {
796
+ w.open(indent + 2, false)
797
+ w.text(it.text!)
798
+ }
799
+ else {
800
+ w.text(' ' + it.text)
801
+ }
802
+ noted = true
803
+ continue
804
+ }
805
+ w.open(indent + 2, false)
806
+ emitValue(w, it, indent + 2)
807
+ if (items.slice(k + 1).some((x) => 'note' !== x.t)) {
808
+ w.text(',')
809
+ }
810
+ noted = false
811
+ }
812
+ w.open(indent, false)
813
+ w.text(')')
814
+ }
815
+
816
+ // A container on several lines (§3.5): the opener ends its line, the
817
+ // entries are statements one level in, the closer stands alone. An
818
+ // empty container is inline whatever the budget says.
819
+ function emitBlock(w: Writer, open: string, close: string, node: Node, indent: number): void {
820
+ if (0 === node.body!.length && undefined === node.open) {
821
+ w.text(open + close)
822
+ return
823
+ }
824
+ w.text(open)
825
+ if (undefined !== node.open) {
826
+ w.text(' ' + node.open)
827
+ }
828
+ emitBody(w, node.body!, indent + 2)
829
+ w.open(indent, false)
830
+ w.text(close)
831
+ }
832
+
833
+ // An expression that has no one-line form, or one too wide for the
834
+ // budget: the author's breaks are kept, each at its operator, which
835
+ // leads its continuation line (§3.11). The continuation is one level
836
+ // in when the expression follows a key on its line, and level with
837
+ // the first operand when the expression has the line to itself -- an
838
+ // argument of a block call, say -- so a disjunction of alternatives
839
+ // reads as the list it is. A container operand that does not fit from
840
+ // where it stands is a block whose closer lines up with the line that
841
+ // opened it.
842
+ function emitExpr(w: Writer, items: Node[], indent: number): void {
843
+ const cont = w.fresh() ? indent : indent + 2
844
+ // Whether the last item was an operand: a comment after one is a
845
+ // space away, and after an operator or the colon it is not. An
846
+ // operand is never directly after an operand (the reader ends a
847
+ // value there), so operands need no such check.
848
+ let operand = false
849
+ let cur = indent
850
+ for (const it of items) {
851
+ if ('op' === it.t) {
852
+ if (it.brk) {
853
+ cur = cont
854
+ if (!w.fresh()) {
855
+ w.open(cur, false)
856
+ }
857
+ w.text(it.text + ' ')
858
+ }
859
+ else {
860
+ w.text(' ' + it.text + ' ')
861
+ }
862
+ operand = false
863
+ continue
864
+ }
865
+ if ('prefix' === it.t) {
866
+ w.text(it.text!)
867
+ operand = false
868
+ continue
869
+ }
870
+ if ('note' === it.t) {
871
+ if (operand) {
872
+ w.text(' ')
873
+ }
874
+ w.text(it.text!)
875
+ cur = cont
876
+ w.open(cur, false)
877
+ operand = false
878
+ continue
879
+ }
880
+ emitValue(w, it, cur)
881
+ operand = true
882
+ }
883
+ }
884
+
885
+ function emit(root: Node[]): string {
886
+ const w = new Writer()
887
+ emitBody(w, root, 0)
888
+ return w.finish()
889
+ }
890
+
891
+
892
+ // ---------------------------------------------------------------------
893
+ // The verb's library surface
894
+
895
+ function lf(text: string): string {
896
+ return text.split('\r\n').join('\n')
897
+ }
898
+
899
+ // The check: the output parses, and to the same tree. Pre-unification
900
+ // canon is that tree, positions aside, and every rewrite of this tier
901
+ // leaves it unchanged (§7.3).
902
+ function sameDocument(root: any, after: string): boolean {
903
+ const p = parseDoc(after, undefined, undefined)
904
+ return undefined === p.errors && root.canon === p.root.canon
905
+ }
906
+
907
+ function depthFinding(): VetFinding {
908
+ return {
909
+ code: 'max_depth',
910
+ class: 'budget',
911
+ severity: 'error',
912
+ path: '$',
913
+ message: `The document nests more than ${MAX_DEPTH} levels deep, past what the formatter reads.`,
914
+ sites: [],
915
+ }
916
+ }
917
+
918
+ function checkFinding(path: string | undefined, expected: string, actual: string): VetFinding {
919
+ return {
920
+ code: 'format_check',
921
+ class: 'internal',
922
+ severity: 'error',
923
+ path: '$',
924
+ message: 'The formatted text is not the same document, so nothing was written.',
925
+ note: 'a formatter defect: please report it with the source' +
926
+ (undefined === path ? '' : ' (' + path + ')'),
927
+ sites: [],
928
+ expected,
929
+ actual,
930
+ }
931
+ }
932
+
933
+ // Format one document. The text is the agreed form of the source;
934
+ // `changed` says whether it differs from what was given, which is
935
+ // what `--check` and `--list` report.
936
+ export function format(src: string, opts?: FormatOptions, hooks?: FormatHooks): FormatReport {
937
+ const text = lf(src)
938
+ const toks: Tok[] = []
939
+ const parsed = parseDoc(text, opts?.path, toks)
940
+ if (undefined !== parsed.errors) {
941
+ return { verdict: 'error', errors: parsed.errors }
942
+ }
943
+ const reader = new Reader(toks)
944
+ const root = reader.body('', false).body
945
+ if (reader.deep) {
946
+ return { verdict: 'error', errors: [depthFinding()] }
947
+ }
948
+ const out = emit(unwrap(root))
949
+ const same = hooks?.same ?? sameDocument
950
+ if (!same(parsed.root, out)) {
951
+ return {
952
+ verdict: 'error',
953
+ errors: [checkFinding(opts?.path, parsed.root.canon, out)],
954
+ }
955
+ }
956
+ return { verdict: 'formatted', text: out, changed: out !== src }
957
+ }
958
+
959
+
960
+ // ---------------------------------------------------------------------
961
+ // The unified diff of `--diff`
962
+
963
+ // A patience diff: lines unique to both sides, in order, are the
964
+ // anchors, and the gaps between them recurse. Not always the shortest
965
+ // edit script, but linear in space, and the same script from both
966
+ // ports, which is what a shared golden needs.
967
+
968
+ type Edit = { op: ' ' | '-' | '+', text: string }
969
+
970
+ // The lines of a text, with a marker on the last when the text does
971
+ // not end in a newline: such a line never equals its
972
+ // newline-terminated twin, which is how the diff reports the
973
+ // difference, and the marker is rendered as diff renders it. NUL,
974
+ // which no source line ends in.
975
+ const NO_NEWLINE = String.fromCharCode(0)
976
+
977
+ function textLines(text: string): string[] {
978
+ if ('' === text) {
979
+ return []
980
+ }
981
+ const lines = text.split('\n')
982
+ if ('' === lines[lines.length - 1]) {
983
+ lines.pop()
984
+ }
985
+ else {
986
+ lines[lines.length - 1] += NO_NEWLINE
987
+ }
988
+ return lines
989
+ }
990
+
991
+ // The longest chain of anchors in order on both sides: patience
992
+ // sorting over the right-hand positions, with the left already
993
+ // ascending.
994
+ function longestChain(pairs: [number, number][]): [number, number][] {
995
+ const tails: number[] = []
996
+ const prev: number[] = []
997
+ for (let k = 0; k < pairs.length; k++) {
998
+ const j = pairs[k][1]
999
+ let lo = 0
1000
+ let hi = tails.length
1001
+ while (lo < hi) {
1002
+ const mid = (lo + hi) >> 1
1003
+ if (pairs[tails[mid]][1] < j) {
1004
+ lo = mid + 1
1005
+ }
1006
+ else {
1007
+ hi = mid
1008
+ }
1009
+ }
1010
+ prev[k] = 0 < lo ? tails[lo - 1] : -1
1011
+ tails[lo] = k
1012
+ }
1013
+ const out: [number, number][] = []
1014
+ let k = 0 === tails.length ? -1 : tails[tails.length - 1]
1015
+ while (0 <= k) {
1016
+ out.push(pairs[k])
1017
+ k = prev[k]
1018
+ }
1019
+ return out.reverse()
1020
+ }
1021
+
1022
+ function patience(
1023
+ a: string[], x0: number, x1: number, b: string[], y0: number, y1: number, out: Edit[]
1024
+ ): void {
1025
+ while (x0 < x1 && y0 < y1 && a[x0] === b[y0]) {
1026
+ out.push({ op: ' ', text: a[x0] })
1027
+ x0++
1028
+ y0++
1029
+ }
1030
+ let tail = 0
1031
+ while (x0 < x1 - tail && y0 < y1 - tail && a[x1 - 1 - tail] === b[y1 - 1 - tail]) {
1032
+ tail++
1033
+ }
1034
+ x1 -= tail
1035
+ y1 -= tail
1036
+
1037
+ const countA = new Map<string, number>()
1038
+ const countB = new Map<string, number>()
1039
+ const posB = new Map<string, number>()
1040
+ for (let x = x0; x < x1; x++) {
1041
+ countA.set(a[x], (countA.get(a[x]) ?? 0) + 1)
1042
+ }
1043
+ for (let y = y0; y < y1; y++) {
1044
+ countB.set(b[y], (countB.get(b[y]) ?? 0) + 1)
1045
+ posB.set(b[y], y)
1046
+ }
1047
+ const pairs: [number, number][] = []
1048
+ for (let x = x0; x < x1; x++) {
1049
+ if (1 === countA.get(a[x]) && 1 === countB.get(a[x])) {
1050
+ pairs.push([x, posB.get(a[x])!])
1051
+ }
1052
+ }
1053
+ const anchors = longestChain(pairs)
1054
+
1055
+ if (0 === anchors.length) {
1056
+ for (let x = x0; x < x1; x++) {
1057
+ out.push({ op: '-', text: a[x] })
1058
+ }
1059
+ for (let y = y0; y < y1; y++) {
1060
+ out.push({ op: '+', text: b[y] })
1061
+ }
1062
+ }
1063
+ else {
1064
+ let x = x0
1065
+ let y = y0
1066
+ for (const [ax, ay] of anchors) {
1067
+ patience(a, x, ax, b, y, ay, out)
1068
+ out.push({ op: ' ', text: a[ax] })
1069
+ x = ax + 1
1070
+ y = ay + 1
1071
+ }
1072
+ patience(a, x, x1, b, y, y1, out)
1073
+ }
1074
+
1075
+ for (let k = 0; k < tail; k++) {
1076
+ out.push({ op: ' ', text: a[x1 + k] })
1077
+ }
1078
+ }
1079
+
1080
+ // The diff in unified format, three lines of context, the file named
1081
+ // on both sides. Empty when the texts are the same.
1082
+ export function unifiedDiff(name: string, before: string, after: string): string {
1083
+ const a = textLines(before)
1084
+ const b = textLines(after)
1085
+ const edits: Edit[] = []
1086
+ patience(a, 0, a.length, b, 0, b.length, edits)
1087
+
1088
+ // Hunks: changes closer than twice the context share one.
1089
+ const hunks: [number, number][] = []
1090
+ for (let k = 0; k < edits.length; k++) {
1091
+ if (' ' === edits[k].op) {
1092
+ continue
1093
+ }
1094
+ const last = hunks[hunks.length - 1]
1095
+ if (undefined !== last && k - last[1] <= 6) {
1096
+ last[1] = k
1097
+ }
1098
+ else {
1099
+ hunks.push([k, k])
1100
+ }
1101
+ }
1102
+ if (0 === hunks.length) {
1103
+ return ''
1104
+ }
1105
+
1106
+ const out: string[] = ['--- a/' + name, '+++ b/' + name]
1107
+ let ai = 0
1108
+ let bi = 0
1109
+ let next = 0
1110
+ for (const [s, e] of hunks) {
1111
+ const from = Math.max(s - 3, 0)
1112
+ const to = Math.min(e + 4, edits.length)
1113
+ // Everything between two hunks is context -- a change would have
1114
+ // opened a hunk -- so both sides advance together.
1115
+ for (; next < from; next++) {
1116
+ ai++
1117
+ bi++
1118
+ }
1119
+ let alen = 0
1120
+ let blen = 0
1121
+ const lines: string[] = []
1122
+ for (let k = from; k < to; k++) {
1123
+ const ed = edits[k]
1124
+ if ('+' !== ed.op) {
1125
+ alen++
1126
+ }
1127
+ if ('-' !== ed.op) {
1128
+ blen++
1129
+ }
1130
+ if (ed.text.endsWith(NO_NEWLINE)) {
1131
+ lines.push(ed.op + ed.text.slice(0, -1))
1132
+ lines.push('\')
1133
+ }
1134
+ else {
1135
+ lines.push(ed.op + ed.text)
1136
+ }
1137
+ }
1138
+ out.push('@@ -' + (0 === alen ? ai : ai + 1) + ',' + alen +
1139
+ ' +' + (0 === blen ? bi : bi + 1) + ',' + blen + ' @@')
1140
+ out.push(...lines)
1141
+ ai += alen
1142
+ bi += blen
1143
+ next = to
1144
+ }
1145
+ return out.join('\n') + '\n'
1146
+ }