@hviana/sema 0.7.5 → 0.7.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/src/mind/graph-search.d.ts +32 -22
- package/dist/src/mind/graph-search.js +97 -54
- package/jsr.json +1 -1
- package/package.json +1 -1
- package/src/mind/graph-search.ts +101 -55
- package/test/98-completion-chaining.test.mjs +140 -0
|
@@ -270,37 +270,47 @@ export declare class GraphSearch {
|
|
|
270
270
|
* and recompose into a deeper learnt form (→ FINAL). This is why a single
|
|
271
271
|
* edge-target needs no bespoke logic — it routes back through {@link solve}.
|
|
272
272
|
*
|
|
273
|
-
*
|
|
274
|
-
*
|
|
275
|
-
*
|
|
273
|
+
* The produced bytes are decomposed by the machinery that owns their shape:
|
|
274
|
+
* the span's LEAVES and SPLITS drive the split rule, which resolves each half
|
|
275
|
+
* through findLeaf — so a part that straddles a content-defined cut is still
|
|
276
|
+
* recovered even though the node's tree children need not align with the
|
|
277
|
+
* learnt parts (the fold cuts "p1 p2" as "p1 p"|"2"). Recognised SITES are
|
|
278
|
+
* filtered to the node's own kids: a produced form is completed out of what
|
|
279
|
+
* it was built from, never by re-recognising arbitrary forms inside it. That
|
|
280
|
+
* filter is load-bearing — a 37-byte dialogue sentence carries seven hub
|
|
281
|
+
* openers, and re-covering them chained through the corpus's whole
|
|
282
|
+
* continuation population: 2.5 GB and OOM for `respond("hi.")`.
|
|
276
283
|
*
|
|
277
284
|
* The recovered answer is accepted only when it MOVED and names a LEARNT node
|
|
278
285
|
* ({@link resolve}) — the graph itself gates against re-expanding a contained
|
|
279
|
-
* form ("ice is cold" ⊅→ "ice is cold is cold").
|
|
286
|
+
* form ("ice is cold" ⊅→ "ice is cold is cold"). An ACCEPTED completion is
|
|
287
|
+
* then re-covered in turn, so a chain runs as deep as the graph licenses; a
|
|
288
|
+
* rejected one ends its branch, so no work is spent past it.
|
|
280
289
|
*
|
|
281
|
-
* Termination
|
|
282
|
-
*
|
|
283
|
-
*
|
|
284
|
-
*
|
|
285
|
-
*
|
|
286
|
-
*
|
|
287
|
-
* and each finished completion is memoised". That is a bound of N — the one
|
|
288
|
-
* AGENTS §2.8 forbids — and it was load-bearing, not pedantic: nested, the
|
|
289
|
-
* recursion reached depth 331 and 9.1 GB on an 18.9M-node store for a 2-byte
|
|
290
|
-
* query and did not terminate, which is what killed a 5 h training run at its
|
|
291
|
-
* checkpoint recall. Guard: test/89-completion-recursion.test.mjs. */
|
|
290
|
+
* Termination and cost are STRUCTURAL: {@link recompleteOpen} is the chain's
|
|
291
|
+
* stack (membership is the cycle guard), {@link recompleteMemo} re-covers each
|
|
292
|
+
* node at most once, only accepted completions recurse, and every level
|
|
293
|
+
* deepens only the top derivation — so a cover pays for the chain it finds,
|
|
294
|
+
* not for how densely the corpus interconnects the forms it passes through.
|
|
295
|
+
* Guard: test/89-completion-recursion.test.mjs. */
|
|
292
296
|
private recompleteNode;
|
|
293
297
|
/** Per-cover memo of each produced node's completion (so the many terminal
|
|
294
298
|
* outs of a long query re-cover each distinct node at most once); reset at the
|
|
295
299
|
* top of {@link cover}. */
|
|
296
300
|
private recompleteMemo;
|
|
297
|
-
/** The
|
|
298
|
-
*
|
|
299
|
-
*
|
|
300
|
-
* is
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
*
|
|
301
|
+
/** The derivation sink of the TOP cover, threaded into every nested
|
|
302
|
+
* completion so a produced form's own recompositions are reported in the
|
|
303
|
+
* same trace instead of vanishing after the first layer. Undefined when
|
|
304
|
+
* nothing is inspecting, so an uninspected response pays nothing. */
|
|
305
|
+
private derivationSink?;
|
|
306
|
+
/** The chain of nodes currently being re-completed — the recursion STACK.
|
|
307
|
+
* MEMBERSHIP is the cycle guard ({@link recompleteNode} refuses a node
|
|
308
|
+
* already open on this chain), which is what lets a completion recurse as
|
|
309
|
+
* deep as the graph licenses while the work stays the answer's: the
|
|
310
|
+
* recursion only advances through an ACCEPTED completion, every produced
|
|
311
|
+
* form is decomposed into its own kids, and the memo re-covers each node at
|
|
312
|
+
* most once per cover. A Set, not a flag, because it states WHICH node is
|
|
313
|
+
* open — the invariant a reader needs to check the guard. */
|
|
304
314
|
private recompleteOpen;
|
|
305
315
|
/** out(i,j,bytes,…): index it for the binary rules, then offer splicing a
|
|
306
316
|
* learnt connector (the in-search bridge), splitting (at a sub-leaf form
|
|
@@ -230,12 +230,26 @@ export class GraphSearch {
|
|
|
230
230
|
// through (completion is cover, recursively — see {@link recompleteNode}).
|
|
231
231
|
this.recompleteOpen.clear();
|
|
232
232
|
this.recompleteMemo = new Map();
|
|
233
|
-
|
|
233
|
+
// The top cover's derivation sink is threaded into every nested completion
|
|
234
|
+
// so the recompositions a produced form needs are reported in the same
|
|
235
|
+
// trace instead of vanishing after the first layer.
|
|
236
|
+
this.derivationSink = onDerivation;
|
|
237
|
+
const solved = this.solve(queryLen, {
|
|
234
238
|
sites,
|
|
235
239
|
leaves,
|
|
236
240
|
splits,
|
|
237
241
|
starts,
|
|
238
242
|
}, conceptTarget, substitutions, connectors, computedResults, onDerivation);
|
|
243
|
+
// Deepening runs HERE, once, on the derivation the top cover CHOSE — never
|
|
244
|
+
// inside the nested solve a completion runs. Nesting the deepening is what
|
|
245
|
+
// made per-query cost track how densely the corpus interconnects the forms
|
|
246
|
+
// passed through: every re-cover fanned out into the whole corpus's
|
|
247
|
+
// continuations instead of following the chain the answer itself licensed.
|
|
248
|
+
// With deepening only at the top, `recompleteNode` walks the accepted chain
|
|
249
|
+
// one link at a time (its own memo and stack), so the work is the answer's.
|
|
250
|
+
return solved === null
|
|
251
|
+
? null
|
|
252
|
+
: { segs: this.deepen(solved.segs), cost: solved.cost };
|
|
239
253
|
}
|
|
240
254
|
/** Build the deduction system for one span and return its lightest cover's
|
|
241
255
|
* chosen spans — the SINGLE routine the query and every produced composite
|
|
@@ -270,7 +284,7 @@ export class GraphSearch {
|
|
|
270
284
|
onDerivation(readDerivation(derivation, substitutions !== undefined));
|
|
271
285
|
}
|
|
272
286
|
return derivation
|
|
273
|
-
? { segs:
|
|
287
|
+
? { segs: readCover(derivation), cost: derivation.cost }
|
|
274
288
|
: null;
|
|
275
289
|
}
|
|
276
290
|
/** Re-cover the CHOSEN fixpoint spans, in place.
|
|
@@ -722,55 +736,51 @@ export class GraphSearch {
|
|
|
722
736
|
* and recompose into a deeper learnt form (→ FINAL). This is why a single
|
|
723
737
|
* edge-target needs no bespoke logic — it routes back through {@link solve}.
|
|
724
738
|
*
|
|
725
|
-
*
|
|
726
|
-
*
|
|
727
|
-
*
|
|
739
|
+
* The produced bytes are decomposed by the machinery that owns their shape:
|
|
740
|
+
* the span's LEAVES and SPLITS drive the split rule, which resolves each half
|
|
741
|
+
* through findLeaf — so a part that straddles a content-defined cut is still
|
|
742
|
+
* recovered even though the node's tree children need not align with the
|
|
743
|
+
* learnt parts (the fold cuts "p1 p2" as "p1 p"|"2"). Recognised SITES are
|
|
744
|
+
* filtered to the node's own kids: a produced form is completed out of what
|
|
745
|
+
* it was built from, never by re-recognising arbitrary forms inside it. That
|
|
746
|
+
* filter is load-bearing — a 37-byte dialogue sentence carries seven hub
|
|
747
|
+
* openers, and re-covering them chained through the corpus's whole
|
|
748
|
+
* continuation population: 2.5 GB and OOM for `respond("hi.")`.
|
|
728
749
|
*
|
|
729
750
|
* The recovered answer is accepted only when it MOVED and names a LEARNT node
|
|
730
751
|
* ({@link resolve}) — the graph itself gates against re-expanding a contained
|
|
731
|
-
* form ("ice is cold" ⊅→ "ice is cold is cold").
|
|
752
|
+
* form ("ice is cold" ⊅→ "ice is cold is cold"). An ACCEPTED completion is
|
|
753
|
+
* then re-covered in turn, so a chain runs as deep as the graph licenses; a
|
|
754
|
+
* rejected one ends its branch, so no work is spent past it.
|
|
732
755
|
*
|
|
733
|
-
* Termination
|
|
734
|
-
*
|
|
735
|
-
*
|
|
736
|
-
*
|
|
737
|
-
*
|
|
738
|
-
*
|
|
739
|
-
* and each finished completion is memoised". That is a bound of N — the one
|
|
740
|
-
* AGENTS §2.8 forbids — and it was load-bearing, not pedantic: nested, the
|
|
741
|
-
* recursion reached depth 331 and 9.1 GB on an 18.9M-node store for a 2-byte
|
|
742
|
-
* query and did not terminate, which is what killed a 5 h training run at its
|
|
743
|
-
* checkpoint recall. Guard: test/89-completion-recursion.test.mjs. */
|
|
756
|
+
* Termination and cost are STRUCTURAL: {@link recompleteOpen} is the chain's
|
|
757
|
+
* stack (membership is the cycle guard), {@link recompleteMemo} re-covers each
|
|
758
|
+
* node at most once, only accepted completions recurse, and every level
|
|
759
|
+
* deepens only the top derivation — so a cover pays for the chain it finds,
|
|
760
|
+
* not for how densely the corpus interconnects the forms it passes through.
|
|
761
|
+
* Guard: test/89-completion-recursion.test.mjs. */
|
|
744
762
|
recompleteNode(node) {
|
|
745
763
|
if (!this.host.recogniseSpan)
|
|
746
764
|
return null;
|
|
747
765
|
const memo = this.recompleteMemo;
|
|
748
766
|
if (memo.has(node))
|
|
749
767
|
return memo.get(node) ?? null;
|
|
750
|
-
// ONE re-cover per produced node — never a re-cover inside a re-cover.
|
|
751
|
-
//
|
|
752
768
|
// Re-covering is how a PRODUCED node's bytes enter the search at all: the
|
|
753
|
-
// cover machinery otherwise only ever sees the QUERY's spans.
|
|
754
|
-
//
|
|
755
|
-
//
|
|
756
|
-
//
|
|
757
|
-
//
|
|
758
|
-
//
|
|
759
|
-
// bytes nothing else brings in) needs it.
|
|
769
|
+
// cover machinery otherwise only ever sees the QUERY's spans. The recursion
|
|
770
|
+
// is allowed to nest — a chain IS nested completions — but it is bounded so
|
|
771
|
+
// the work stays the ANSWER's (AGENTS §2.8): the stack below is the cycle
|
|
772
|
+
// guard, only ACCEPTED completions recurse, and the nested solve decomposes
|
|
773
|
+
// the form by its own shape instead of re-recognising the corpus's hub forms
|
|
774
|
+
// inside it.
|
|
760
775
|
//
|
|
761
|
-
//
|
|
762
|
-
//
|
|
763
|
-
//
|
|
764
|
-
//
|
|
765
|
-
//
|
|
766
|
-
//
|
|
767
|
-
//
|
|
768
|
-
|
|
769
|
-
//
|
|
770
|
-
// `recompleteOpen` is that stack, so a non-empty stack means we are already
|
|
771
|
-
// inside one. This subsumes the old cycle guard: a node cannot recurse
|
|
772
|
-
// back into itself when nothing recurses at all.
|
|
773
|
-
if (this.recompleteOpen.size > 0)
|
|
776
|
+
// `recompleteOpen` IS the stack of the chain being built, so MEMBERSHIP is
|
|
777
|
+
// the cycle guard: a node already open on this chain cannot re-enter it.
|
|
778
|
+
// Testing the NODE — not the stack's size — is what lets a completion
|
|
779
|
+
// recurse as deep as the graph licenses, exactly the intrinsic convergence
|
|
780
|
+
// {@link solve}'s contract states. Work stays the answer's because the
|
|
781
|
+
// recursion only ever advances through an ACCEPTED completion (below) and
|
|
782
|
+
// {@link cover} deepens only the top derivation.
|
|
783
|
+
if (this.recompleteOpen.has(node))
|
|
774
784
|
return null;
|
|
775
785
|
// A leaf or single-child node has no parts to recompose; skip before the
|
|
776
786
|
// costly recognition so a plain terminal answer pays nothing.
|
|
@@ -782,16 +792,43 @@ export class GraphSearch {
|
|
|
782
792
|
const bytes = this.store.bytesPrefix(node, ALL);
|
|
783
793
|
this.recompleteOpen.add(node);
|
|
784
794
|
try {
|
|
785
|
-
// Completion is cover
|
|
786
|
-
//
|
|
787
|
-
//
|
|
788
|
-
//
|
|
789
|
-
|
|
795
|
+
// Completion is cover, but the produced form is decomposed by its own
|
|
796
|
+
// shape. The LEAVES and SPLITS are kept whole, so the split rule still
|
|
797
|
+
// recovers a part that straddles a content-defined cut (the fold cuts
|
|
798
|
+
// "p1 p2" as "p1 p"|"2", and findLeaf still resolves p1 and p2). The
|
|
799
|
+
// recognised SITES are filtered to the node's own kids, because
|
|
800
|
+
// re-recognising arbitrary forms inside the bytes is what let a hub-heavy
|
|
801
|
+
// utterance explode: a 37-byte dialogue sentence carries seven hub
|
|
802
|
+
// openers, and re-covering them chained through the corpus's whole
|
|
803
|
+
// continuation population (2.5 GB and OOM for `"hi."`). No
|
|
804
|
+
// concepts/connectors either (those need the caller's async
|
|
805
|
+
// pre-resolution) — the recursion follows edges and fusion, which is what
|
|
806
|
+
// a deeper rewrite chain is made of.
|
|
807
|
+
const rec = this.host.recogniseSpan(bytes);
|
|
808
|
+
const kids = new Set(nrec.kids);
|
|
809
|
+
const solved = this.solve(bytes.length, {
|
|
810
|
+
sites: rec.sites.filter((s) => kids.has(s.payload)),
|
|
811
|
+
leaves: rec.leaves,
|
|
812
|
+
splits: rec.splits,
|
|
813
|
+
starts: rec.starts,
|
|
814
|
+
}, new Map(), undefined, undefined, undefined, this.derivationSink);
|
|
790
815
|
const answer = solved && concatBytes(solved.segs.map((s) => s.bytes));
|
|
791
|
-
|
|
792
|
-
|
|
793
|
-
|
|
816
|
+
// ACCEPT, then CONTINUE THE CHAIN — but only along an accepted
|
|
817
|
+
// completion. A re-cover whose result is not itself a learnt node (the
|
|
818
|
+
// 70→374-byte concatenations that name nothing) ends its branch here
|
|
819
|
+
// instead of recursing into work the answer never asked for, and an
|
|
820
|
+
// accepted one names a node that is re-covered in turn. That is what
|
|
821
|
+
// keeps a deep chain possible while the work stays proportional to the
|
|
822
|
+
// chain rather than to the corpus's interconnections.
|
|
823
|
+
const composed = answer !== null && !bytesEqual(answer, bytes)
|
|
824
|
+
? this.host.resolve(answer)
|
|
794
825
|
: null;
|
|
826
|
+
if (composed === null) {
|
|
827
|
+
memo.set(node, null);
|
|
828
|
+
return null;
|
|
829
|
+
}
|
|
830
|
+
const deeper = this.recompleteNode(composed);
|
|
831
|
+
const out = deeper ?? answer;
|
|
795
832
|
memo.set(node, out);
|
|
796
833
|
return out;
|
|
797
834
|
}
|
|
@@ -803,13 +840,19 @@ export class GraphSearch {
|
|
|
803
840
|
* outs of a long query re-cover each distinct node at most once); reset at the
|
|
804
841
|
* top of {@link cover}. */
|
|
805
842
|
recompleteMemo = new Map();
|
|
806
|
-
/** The
|
|
807
|
-
*
|
|
808
|
-
*
|
|
809
|
-
* is
|
|
810
|
-
|
|
811
|
-
|
|
812
|
-
*
|
|
843
|
+
/** The derivation sink of the TOP cover, threaded into every nested
|
|
844
|
+
* completion so a produced form's own recompositions are reported in the
|
|
845
|
+
* same trace instead of vanishing after the first layer. Undefined when
|
|
846
|
+
* nothing is inspecting, so an uninspected response pays nothing. */
|
|
847
|
+
derivationSink;
|
|
848
|
+
/** The chain of nodes currently being re-completed — the recursion STACK.
|
|
849
|
+
* MEMBERSHIP is the cycle guard ({@link recompleteNode} refuses a node
|
|
850
|
+
* already open on this chain), which is what lets a completion recurse as
|
|
851
|
+
* deep as the graph licenses while the work stays the answer's: the
|
|
852
|
+
* recursion only advances through an ACCEPTED completion, every produced
|
|
853
|
+
* form is decomposed into its own kids, and the memo re-covers each node at
|
|
854
|
+
* most once per cover. A Set, not a flag, because it states WHICH node is
|
|
855
|
+
* open — the invariant a reader needs to check the guard. */
|
|
813
856
|
recompleteOpen = new Set();
|
|
814
857
|
/** out(i,j,bytes,…): index it for the binary rules, then offer splicing a
|
|
815
858
|
* learnt connector (the in-search bridge), splitting (at a sub-leaf form
|
package/jsr.json
CHANGED
package/package.json
CHANGED
package/src/mind/graph-search.ts
CHANGED
|
@@ -412,7 +412,11 @@ export class GraphSearch {
|
|
|
412
412
|
// through (completion is cover, recursively — see {@link recompleteNode}).
|
|
413
413
|
this.recompleteOpen.clear();
|
|
414
414
|
this.recompleteMemo = new Map<number, Uint8Array | null>();
|
|
415
|
-
|
|
415
|
+
// The top cover's derivation sink is threaded into every nested completion
|
|
416
|
+
// so the recompositions a produced form needs are reported in the same
|
|
417
|
+
// trace instead of vanishing after the first layer.
|
|
418
|
+
this.derivationSink = onDerivation;
|
|
419
|
+
const solved = this.solve(
|
|
416
420
|
queryLen,
|
|
417
421
|
{
|
|
418
422
|
sites,
|
|
@@ -426,8 +430,17 @@ export class GraphSearch {
|
|
|
426
430
|
computedResults,
|
|
427
431
|
onDerivation,
|
|
428
432
|
);
|
|
433
|
+
// Deepening runs HERE, once, on the derivation the top cover CHOSE — never
|
|
434
|
+
// inside the nested solve a completion runs. Nesting the deepening is what
|
|
435
|
+
// made per-query cost track how densely the corpus interconnects the forms
|
|
436
|
+
// passed through: every re-cover fanned out into the whole corpus's
|
|
437
|
+
// continuations instead of following the chain the answer itself licensed.
|
|
438
|
+
// With deepening only at the top, `recompleteNode` walks the accepted chain
|
|
439
|
+
// one link at a time (its own memo and stack), so the work is the answer's.
|
|
440
|
+
return solved === null
|
|
441
|
+
? null
|
|
442
|
+
: { segs: this.deepen(solved.segs), cost: solved.cost };
|
|
429
443
|
}
|
|
430
|
-
|
|
431
444
|
/** Build the deduction system for one span and return its lightest cover's
|
|
432
445
|
* chosen spans — the SINGLE routine the query and every produced composite
|
|
433
446
|
* run through. `recognition` carries the span's recognised forms; the query
|
|
@@ -484,7 +497,7 @@ export class GraphSearch {
|
|
|
484
497
|
onDerivation(readDerivation(derivation, substitutions !== undefined));
|
|
485
498
|
}
|
|
486
499
|
return derivation
|
|
487
|
-
? { segs:
|
|
500
|
+
? { segs: readCover(derivation), cost: derivation.cost }
|
|
488
501
|
: null;
|
|
489
502
|
}
|
|
490
503
|
|
|
@@ -974,53 +987,49 @@ export class GraphSearch {
|
|
|
974
987
|
* and recompose into a deeper learnt form (→ FINAL). This is why a single
|
|
975
988
|
* edge-target needs no bespoke logic — it routes back through {@link solve}.
|
|
976
989
|
*
|
|
977
|
-
*
|
|
978
|
-
*
|
|
979
|
-
*
|
|
990
|
+
* The produced bytes are decomposed by the machinery that owns their shape:
|
|
991
|
+
* the span's LEAVES and SPLITS drive the split rule, which resolves each half
|
|
992
|
+
* through findLeaf — so a part that straddles a content-defined cut is still
|
|
993
|
+
* recovered even though the node's tree children need not align with the
|
|
994
|
+
* learnt parts (the fold cuts "p1 p2" as "p1 p"|"2"). Recognised SITES are
|
|
995
|
+
* filtered to the node's own kids: a produced form is completed out of what
|
|
996
|
+
* it was built from, never by re-recognising arbitrary forms inside it. That
|
|
997
|
+
* filter is load-bearing — a 37-byte dialogue sentence carries seven hub
|
|
998
|
+
* openers, and re-covering them chained through the corpus's whole
|
|
999
|
+
* continuation population: 2.5 GB and OOM for `respond("hi.")`.
|
|
980
1000
|
*
|
|
981
1001
|
* The recovered answer is accepted only when it MOVED and names a LEARNT node
|
|
982
1002
|
* ({@link resolve}) — the graph itself gates against re-expanding a contained
|
|
983
|
-
* form ("ice is cold" ⊅→ "ice is cold is cold").
|
|
984
|
-
*
|
|
985
|
-
*
|
|
986
|
-
* another re-cover (see the guard below), so one cover pays at most one
|
|
987
|
-
* nested {@link solve} per distinct produced node it actually reaches, and
|
|
988
|
-
* {@link recompleteMemo} collapses a repeat to nothing.
|
|
1003
|
+
* form ("ice is cold" ⊅→ "ice is cold is cold"). An ACCEPTED completion is
|
|
1004
|
+
* then re-covered in turn, so a chain runs as deep as the graph licenses; a
|
|
1005
|
+
* rejected one ends its branch, so no work is spent past it.
|
|
989
1006
|
*
|
|
990
|
-
*
|
|
991
|
-
*
|
|
992
|
-
*
|
|
993
|
-
*
|
|
994
|
-
*
|
|
995
|
-
*
|
|
1007
|
+
* Termination and cost are STRUCTURAL: {@link recompleteOpen} is the chain's
|
|
1008
|
+
* stack (membership is the cycle guard), {@link recompleteMemo} re-covers each
|
|
1009
|
+
* node at most once, only accepted completions recurse, and every level
|
|
1010
|
+
* deepens only the top derivation — so a cover pays for the chain it finds,
|
|
1011
|
+
* not for how densely the corpus interconnects the forms it passes through.
|
|
1012
|
+
* Guard: test/89-completion-recursion.test.mjs. */
|
|
996
1013
|
private recompleteNode(node: number): Uint8Array | null {
|
|
997
1014
|
if (!this.host.recogniseSpan) return null;
|
|
998
1015
|
const memo = this.recompleteMemo;
|
|
999
1016
|
if (memo.has(node)) return memo.get(node) ?? null;
|
|
1000
|
-
// ONE re-cover per produced node — never a re-cover inside a re-cover.
|
|
1001
|
-
//
|
|
1002
1017
|
// Re-covering is how a PRODUCED node's bytes enter the search at all: the
|
|
1003
|
-
// cover machinery otherwise only ever sees the QUERY's spans.
|
|
1004
|
-
//
|
|
1005
|
-
//
|
|
1006
|
-
//
|
|
1007
|
-
//
|
|
1008
|
-
//
|
|
1009
|
-
// bytes nothing else brings in) needs it.
|
|
1018
|
+
// cover machinery otherwise only ever sees the QUERY's spans. The recursion
|
|
1019
|
+
// is allowed to nest — a chain IS nested completions — but it is bounded so
|
|
1020
|
+
// the work stays the ANSWER's (AGENTS §2.8): the stack below is the cycle
|
|
1021
|
+
// guard, only ACCEPTED completions recurse, and the nested solve decomposes
|
|
1022
|
+
// the form by its own shape instead of re-recognising the corpus's hub forms
|
|
1023
|
+
// inside it.
|
|
1010
1024
|
//
|
|
1011
|
-
//
|
|
1012
|
-
//
|
|
1013
|
-
//
|
|
1014
|
-
//
|
|
1015
|
-
//
|
|
1016
|
-
//
|
|
1017
|
-
//
|
|
1018
|
-
|
|
1019
|
-
//
|
|
1020
|
-
// `recompleteOpen` is that stack, so a non-empty stack means we are already
|
|
1021
|
-
// inside one. This subsumes the old cycle guard: a node cannot recurse
|
|
1022
|
-
// back into itself when nothing recurses at all.
|
|
1023
|
-
if (this.recompleteOpen.size > 0) return null;
|
|
1025
|
+
// `recompleteOpen` IS the stack of the chain being built, so MEMBERSHIP is
|
|
1026
|
+
// the cycle guard: a node already open on this chain cannot re-enter it.
|
|
1027
|
+
// Testing the NODE — not the stack's size — is what lets a completion
|
|
1028
|
+
// recurse as deep as the graph licenses, exactly the intrinsic convergence
|
|
1029
|
+
// {@link solve}'s contract states. Work stays the answer's because the
|
|
1030
|
+
// recursion only ever advances through an ACCEPTED completion (below) and
|
|
1031
|
+
// {@link cover} deepens only the top derivation.
|
|
1032
|
+
if (this.recompleteOpen.has(node)) return null;
|
|
1024
1033
|
|
|
1025
1034
|
// A leaf or single-child node has no parts to recompose; skip before the
|
|
1026
1035
|
// costly recognition so a plain terminal answer pays nothing.
|
|
@@ -1033,20 +1042,51 @@ export class GraphSearch {
|
|
|
1033
1042
|
const bytes = this.store.bytesPrefix(node, ALL);
|
|
1034
1043
|
this.recompleteOpen.add(node);
|
|
1035
1044
|
try {
|
|
1036
|
-
// Completion is cover
|
|
1037
|
-
//
|
|
1038
|
-
//
|
|
1039
|
-
//
|
|
1045
|
+
// Completion is cover, but the produced form is decomposed by its own
|
|
1046
|
+
// shape. The LEAVES and SPLITS are kept whole, so the split rule still
|
|
1047
|
+
// recovers a part that straddles a content-defined cut (the fold cuts
|
|
1048
|
+
// "p1 p2" as "p1 p"|"2", and findLeaf still resolves p1 and p2). The
|
|
1049
|
+
// recognised SITES are filtered to the node's own kids, because
|
|
1050
|
+
// re-recognising arbitrary forms inside the bytes is what let a hub-heavy
|
|
1051
|
+
// utterance explode: a 37-byte dialogue sentence carries seven hub
|
|
1052
|
+
// openers, and re-covering them chained through the corpus's whole
|
|
1053
|
+
// continuation population (2.5 GB and OOM for `"hi."`). No
|
|
1054
|
+
// concepts/connectors either (those need the caller's async
|
|
1055
|
+
// pre-resolution) — the recursion follows edges and fusion, which is what
|
|
1056
|
+
// a deeper rewrite chain is made of.
|
|
1057
|
+
const rec = this.host.recogniseSpan(bytes);
|
|
1058
|
+
const kids = new Set(nrec.kids);
|
|
1040
1059
|
const solved = this.solve(
|
|
1041
1060
|
bytes.length,
|
|
1042
|
-
|
|
1061
|
+
{
|
|
1062
|
+
sites: rec.sites.filter((s) => kids.has(s.payload)),
|
|
1063
|
+
leaves: rec.leaves,
|
|
1064
|
+
splits: rec.splits,
|
|
1065
|
+
starts: rec.starts,
|
|
1066
|
+
},
|
|
1043
1067
|
new Map(),
|
|
1068
|
+
undefined,
|
|
1069
|
+
undefined,
|
|
1070
|
+
undefined,
|
|
1071
|
+
this.derivationSink,
|
|
1044
1072
|
);
|
|
1045
1073
|
const answer = solved && concatBytes(solved.segs.map((s) => s.bytes));
|
|
1046
|
-
|
|
1047
|
-
|
|
1048
|
-
|
|
1074
|
+
// ACCEPT, then CONTINUE THE CHAIN — but only along an accepted
|
|
1075
|
+
// completion. A re-cover whose result is not itself a learnt node (the
|
|
1076
|
+
// 70→374-byte concatenations that name nothing) ends its branch here
|
|
1077
|
+
// instead of recursing into work the answer never asked for, and an
|
|
1078
|
+
// accepted one names a node that is re-covered in turn. That is what
|
|
1079
|
+
// keeps a deep chain possible while the work stays proportional to the
|
|
1080
|
+
// chain rather than to the corpus's interconnections.
|
|
1081
|
+
const composed = answer !== null && !bytesEqual(answer, bytes)
|
|
1082
|
+
? this.host.resolve(answer)
|
|
1049
1083
|
: null;
|
|
1084
|
+
if (composed === null) {
|
|
1085
|
+
memo.set(node, null);
|
|
1086
|
+
return null;
|
|
1087
|
+
}
|
|
1088
|
+
const deeper = this.recompleteNode(composed);
|
|
1089
|
+
const out = deeper ?? answer!;
|
|
1050
1090
|
memo.set(node, out);
|
|
1051
1091
|
return out;
|
|
1052
1092
|
} finally {
|
|
@@ -1058,13 +1098,19 @@ export class GraphSearch {
|
|
|
1058
1098
|
* outs of a long query re-cover each distinct node at most once); reset at the
|
|
1059
1099
|
* top of {@link cover}. */
|
|
1060
1100
|
private recompleteMemo = new Map<number, Uint8Array | null>();
|
|
1061
|
-
/** The
|
|
1062
|
-
*
|
|
1063
|
-
*
|
|
1064
|
-
* is
|
|
1065
|
-
|
|
1066
|
-
|
|
1067
|
-
*
|
|
1101
|
+
/** The derivation sink of the TOP cover, threaded into every nested
|
|
1102
|
+
* completion so a produced form's own recompositions are reported in the
|
|
1103
|
+
* same trace instead of vanishing after the first layer. Undefined when
|
|
1104
|
+
* nothing is inspecting, so an uninspected response pays nothing. */
|
|
1105
|
+
private derivationSink?: (steps: DerivationStep[]) => void;
|
|
1106
|
+
/** The chain of nodes currently being re-completed — the recursion STACK.
|
|
1107
|
+
* MEMBERSHIP is the cycle guard ({@link recompleteNode} refuses a node
|
|
1108
|
+
* already open on this chain), which is what lets a completion recurse as
|
|
1109
|
+
* deep as the graph licenses while the work stays the answer's: the
|
|
1110
|
+
* recursion only advances through an ACCEPTED completion, every produced
|
|
1111
|
+
* form is decomposed into its own kids, and the memo re-covers each node at
|
|
1112
|
+
* most once per cover. A Set, not a flag, because it states WHICH node is
|
|
1113
|
+
* open — the invariant a reader needs to check the guard. */
|
|
1068
1114
|
private recompleteOpen = new Set<number>();
|
|
1069
1115
|
|
|
1070
1116
|
/** out(i,j,bytes,…): index it for the binary rules, then offer splicing a
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
// 98-completion-chaining.test.mjs — the completion recursion NESTS, and the
|
|
2
|
+
// two properties that make that reachable are pinned here.
|
|
3
|
+
//
|
|
4
|
+
// `graph-search.ts`'s `recompleteNode` used to refuse any re-cover inside a
|
|
5
|
+
// re-cover (`recompleteOpen.size > 0`), so a chain through a PRODUCED composite
|
|
6
|
+
// stopped after a single recomposition: the depth the A*LD substrate licenses
|
|
7
|
+
// was unreachable. The recursion now nests, but only through ACCEPTED
|
|
8
|
+
// completions and only along the produced form's own parts — and with per-query
|
|
9
|
+
// work that stays the answer's (test/89 pins the cost).
|
|
10
|
+
//
|
|
11
|
+
// WHY THESE AND NOT test/15's §9–§12: those exercise chains whose parts all live
|
|
12
|
+
// INSIDE the query's own span ("a e", "x y", "p q r"); they pass even with
|
|
13
|
+
// `recompleteNode` disabled outright. §6 reaches a produced composite, but only
|
|
14
|
+
// one recomposition deep. The cases here fail on the pre-change tree:
|
|
15
|
+
// 1. the chain stops at the intermediate composite ("m n", not "z"), and
|
|
16
|
+
// 2. the nested derivation never reaches the rationale.
|
|
17
|
+
//
|
|
18
|
+
// MEASURED on the pre-change tree (this file's fixtures):
|
|
19
|
+
// respondText("seed") === "m n" (one recomposition)
|
|
20
|
+
// rationale moves = no `fuse`, no `recompose`
|
|
21
|
+
// After the change: "z", and both moves present.
|
|
22
|
+
|
|
23
|
+
import { test } from "node:test";
|
|
24
|
+
import assert from "node:assert/strict";
|
|
25
|
+
import { Mind } from "../dist/src/index.js";
|
|
26
|
+
|
|
27
|
+
/** seed → "p q" → (p→r, q→s) → "r s" → "m n" → (m→x, n→y) → "x y" → z
|
|
28
|
+
*
|
|
29
|
+
* The query names only `seed`. Every form after the first hop is PRODUCED by
|
|
30
|
+
* an edge, so reaching `z` requires TWO nested completions: "p q" recomposes to
|
|
31
|
+
* "m n", and "m n" recomposes again to "x y" → z. A single-level completion
|
|
32
|
+
* stops one composite short, at "m n". */
|
|
33
|
+
async function deepChain() {
|
|
34
|
+
const m = new Mind({ seed: 7 });
|
|
35
|
+
await m.ingest([
|
|
36
|
+
["seed", "p q"],
|
|
37
|
+
["p", "r"],
|
|
38
|
+
["q", "s"],
|
|
39
|
+
["r s", "m n"],
|
|
40
|
+
["m", "x"],
|
|
41
|
+
["n", "y"],
|
|
42
|
+
["x y", "z"],
|
|
43
|
+
]);
|
|
44
|
+
return m;
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
test("a produced composite completes through two nested recompositions", async () => {
|
|
48
|
+
const m = await deepChain();
|
|
49
|
+
assert.equal(
|
|
50
|
+
(await m.respondText("seed")).replace(/\0+/g, ""),
|
|
51
|
+
"z",
|
|
52
|
+
);
|
|
53
|
+
await m.store.close();
|
|
54
|
+
});
|
|
55
|
+
|
|
56
|
+
test("nested completions reach the rationale", async () => {
|
|
57
|
+
const m = await deepChain();
|
|
58
|
+
const steps = [];
|
|
59
|
+
await m.respondText("seed", (s) => steps.push(s));
|
|
60
|
+
const moves = new Set(steps.map((s) => s.mechanism.at(-1)));
|
|
61
|
+
// The top cover cannot fabricate these on its own: recomposition happens
|
|
62
|
+
// inside the nested solves, and they were invisible before the sink was
|
|
63
|
+
// threaded through.
|
|
64
|
+
assert.ok(
|
|
65
|
+
moves.has("recompose"),
|
|
66
|
+
`expected a recompose move in the rationale, got: ${[...moves].join(", ")}`,
|
|
67
|
+
);
|
|
68
|
+
assert.ok(
|
|
69
|
+
moves.has("fuse"),
|
|
70
|
+
`expected a fuse move in the rationale, got: ${[...moves].join(", ")}`,
|
|
71
|
+
);
|
|
72
|
+
await m.store.close();
|
|
73
|
+
});
|
|
74
|
+
|
|
75
|
+
test("a produced composite is completed by its own parts, not by a learned form inside it", async () => {
|
|
76
|
+
// The produced bytes are "hi there"; `hi` is a learned context that is NOT one
|
|
77
|
+
// of their parts. Re-recognising arbitrary forms inside the bytes is what
|
|
78
|
+
// let a hub-heavy utterance explode on the trained store (`respond("hi.")`:
|
|
79
|
+
// 37 bytes, seven hub openers, 2.5 GB, OOM) — the produced form is decomposed
|
|
80
|
+
// instead by the machinery that owns its shape (leaves/splits), with the
|
|
81
|
+
// recognised SITES filtered to the node's own kids.
|
|
82
|
+
//
|
|
83
|
+
// MEASURED on the pre-change tree: no `recompose` move here at all — the
|
|
84
|
+
// unrelated `hi` site was taken instead of decomposing the form.
|
|
85
|
+
const m = new Mind({ seed: 7 });
|
|
86
|
+
await m.ingest([
|
|
87
|
+
["seed", "hi there"],
|
|
88
|
+
["hi", "KLX"],
|
|
89
|
+
]);
|
|
90
|
+
const steps = [];
|
|
91
|
+
const answer = (await m.respondText("seed", (s) => steps.push(s)))
|
|
92
|
+
.replace(/\0+/g, "")
|
|
93
|
+
.trim();
|
|
94
|
+
assert.equal(answer, "hi there");
|
|
95
|
+
const moves = new Set(steps.map((s) => s.mechanism.at(-1)));
|
|
96
|
+
assert.ok(
|
|
97
|
+
moves.has("recompose"),
|
|
98
|
+
`expected the completion to decompose the form by its parts, got: ${
|
|
99
|
+
[...moves].join(", ")
|
|
100
|
+
}`,
|
|
101
|
+
);
|
|
102
|
+
await m.store.close();
|
|
103
|
+
});
|
|
104
|
+
|
|
105
|
+
test("the chain runs as deep as the graph licenses (four nested completions)", async () => {
|
|
106
|
+
const m = new Mind({ seed: 7 });
|
|
107
|
+
await m.ingest([
|
|
108
|
+
["seed", "p1 q1"],
|
|
109
|
+
["p1", "a1"],
|
|
110
|
+
["q1", "b1"],
|
|
111
|
+
["a1 b1", "p2 q2"],
|
|
112
|
+
["p2", "a2"],
|
|
113
|
+
["q2", "b2"],
|
|
114
|
+
["a2 b2", "p3 q3"],
|
|
115
|
+
["p3", "a3"],
|
|
116
|
+
["q3", "b3"],
|
|
117
|
+
["a3 b3", "FIM"],
|
|
118
|
+
]);
|
|
119
|
+
assert.equal((await m.respondText("seed")).replace(/\0+/g, ""), "FIM");
|
|
120
|
+
await m.store.close();
|
|
121
|
+
});
|
|
122
|
+
|
|
123
|
+
test("a completion cycle terminates deterministically (the stack is the guard)", async () => {
|
|
124
|
+
// The recomposition of "x y" leads back to "x y" itself. Membership in
|
|
125
|
+
// `recompleteOpen` refuses the re-entry, so the recursion terminates on the
|
|
126
|
+
// form it already holds instead of re-covering it forever — and it answers
|
|
127
|
+
// the same bytes twice.
|
|
128
|
+
const m = new Mind({ seed: 7 });
|
|
129
|
+
await m.ingest([
|
|
130
|
+
["seed", "x y"],
|
|
131
|
+
["x", "A"],
|
|
132
|
+
["y", "B"],
|
|
133
|
+
["A B", "x y"],
|
|
134
|
+
]);
|
|
135
|
+
const first = (await m.respondText("seed")).replace(/\0+/g, "").trim();
|
|
136
|
+
const second = (await m.respondText("seed")).replace(/\0+/g, "").trim();
|
|
137
|
+
assert.equal(first, second);
|
|
138
|
+
assert.equal(first, "x y");
|
|
139
|
+
await m.store.close();
|
|
140
|
+
});
|