@hviana/sema 0.7.5 → 0.7.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/src/mind/graph-search.d.ts +56 -25
- package/dist/src/mind/graph-search.js +177 -59
- package/dist/src/mind/recognition.js +35 -1
- package/dist/src/mind/trace.js +1 -0
- package/jsr.json +1 -1
- package/package.json +1 -1
- package/src/mind/graph-search.ts +194 -59
- package/src/mind/recognition.ts +33 -1
- package/src/mind/trace.ts +2 -0
- package/test/46-recognise-multibyte-edge.test.mjs +30 -0
- package/test/98-completion-chaining.test.mjs +140 -0
- package/test/99-fact-join.test.mjs +97 -0
|
@@ -73,6 +73,12 @@ export type GItem = {
|
|
|
73
73
|
* derivation actually CHOSE. Part of {@link key}, because it decides
|
|
74
74
|
* whether the span's final bytes may still change. */
|
|
75
75
|
fix?: boolean;
|
|
76
|
+
/** Set on the out a JOIN produced: a produced fact's own contained entity
|
|
77
|
+
* (the subject the query never named) combined with the query's adjacent
|
|
78
|
+
* relation span named a learned key, and that key's continuation is this
|
|
79
|
+
* out. Part of {@link key} so the joined reading is a distinct chart item
|
|
80
|
+
* from the plain concatenation of the same bytes. */
|
|
81
|
+
join?: boolean;
|
|
76
82
|
};
|
|
77
83
|
export declare const STEP = 1;
|
|
78
84
|
export declare const CONCEPT = 10;
|
|
@@ -136,7 +142,7 @@ export interface DerivationItem {
|
|
|
136
142
|
* {@link GraphSearch}'s rules fired, recovered from the rule's premise/
|
|
137
143
|
* conclusion shape (the rules carry no label, so this classifies by structure,
|
|
138
144
|
* the single place that maps rule geometry to a name). */
|
|
139
|
-
export type DerivationMove = "axiom" | "follow-edge" | "concept-hop" | "voice" | "ground" | "splice-connector" | "split" | "fuse" | "recompose" | "bridge" | "pool-vote" | "step";
|
|
145
|
+
export type DerivationMove = "axiom" | "follow-edge" | "concept-hop" | "voice" | "ground" | "splice-connector" | "split" | "fuse" | "recompose" | "join-fact" | "bridge" | "pool-vote" | "step";
|
|
140
146
|
/** The lightest-derivation search over the Sema graph. One instance binds the
|
|
141
147
|
* store, `maxGroup` (the fusible span ceiling), and the canonical
|
|
142
148
|
* {@link resolve} callback; {@link cover} then solves one query. */
|
|
@@ -270,42 +276,67 @@ export declare class GraphSearch {
|
|
|
270
276
|
* and recompose into a deeper learnt form (→ FINAL). This is why a single
|
|
271
277
|
* edge-target needs no bespoke logic — it routes back through {@link solve}.
|
|
272
278
|
*
|
|
273
|
-
*
|
|
274
|
-
*
|
|
275
|
-
*
|
|
279
|
+
* The produced bytes are decomposed by the machinery that owns their shape:
|
|
280
|
+
* the span's LEAVES and SPLITS drive the split rule, which resolves each half
|
|
281
|
+
* through findLeaf — so a part that straddles a content-defined cut is still
|
|
282
|
+
* recovered even though the node's tree children need not align with the
|
|
283
|
+
* learnt parts (the fold cuts "p1 p2" as "p1 p"|"2"). Recognised SITES are
|
|
284
|
+
* filtered to the node's own kids: a produced form is completed out of what
|
|
285
|
+
* it was built from, never by re-recognising arbitrary forms inside it. That
|
|
286
|
+
* filter is load-bearing — a 37-byte dialogue sentence carries seven hub
|
|
287
|
+
* openers, and re-covering them chained through the corpus's whole
|
|
288
|
+
* continuation population: 2.5 GB and OOM for `respond("hi.")`.
|
|
276
289
|
*
|
|
277
290
|
* The recovered answer is accepted only when it MOVED and names a LEARNT node
|
|
278
291
|
* ({@link resolve}) — the graph itself gates against re-expanding a contained
|
|
279
|
-
* form ("ice is cold" ⊅→ "ice is cold is cold").
|
|
292
|
+
* form ("ice is cold" ⊅→ "ice is cold is cold"). An ACCEPTED completion is
|
|
293
|
+
* then re-covered in turn, so a chain runs as deep as the graph licenses; a
|
|
294
|
+
* rejected one ends its branch, so no work is spent past it.
|
|
280
295
|
*
|
|
281
|
-
* Termination
|
|
282
|
-
*
|
|
283
|
-
*
|
|
284
|
-
*
|
|
285
|
-
*
|
|
286
|
-
*
|
|
287
|
-
* and each finished completion is memoised". That is a bound of N — the one
|
|
288
|
-
* AGENTS §2.8 forbids — and it was load-bearing, not pedantic: nested, the
|
|
289
|
-
* recursion reached depth 331 and 9.1 GB on an 18.9M-node store for a 2-byte
|
|
290
|
-
* query and did not terminate, which is what killed a 5 h training run at its
|
|
291
|
-
* checkpoint recall. Guard: test/89-completion-recursion.test.mjs. */
|
|
296
|
+
* Termination and cost are STRUCTURAL: {@link recompleteOpen} is the chain's
|
|
297
|
+
* stack (membership is the cycle guard), {@link recompleteMemo} re-covers each
|
|
298
|
+
* node at most once, only accepted completions recurse, and every level
|
|
299
|
+
* deepens only the top derivation — so a cover pays for the chain it finds,
|
|
300
|
+
* not for how densely the corpus interconnects the forms it passes through.
|
|
301
|
+
* Guard: test/89-completion-recursion.test.mjs. */
|
|
292
302
|
private recompleteNode;
|
|
293
303
|
/** Per-cover memo of each produced node's completion (so the many terminal
|
|
294
304
|
* outs of a long query re-cover each distinct node at most once); reset at the
|
|
295
305
|
* top of {@link cover}. */
|
|
296
306
|
private recompleteMemo;
|
|
297
|
-
/** The
|
|
298
|
-
*
|
|
299
|
-
*
|
|
300
|
-
* is
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
*
|
|
307
|
+
/** The derivation sink of the TOP cover, threaded into every nested
|
|
308
|
+
* completion so a produced form's own recompositions are reported in the
|
|
309
|
+
* same trace instead of vanishing after the first layer. Undefined when
|
|
310
|
+
* nothing is inspecting, so an uninspected response pays nothing. */
|
|
311
|
+
private derivationSink?;
|
|
312
|
+
/** The chain of nodes currently being re-completed — the recursion STACK.
|
|
313
|
+
* MEMBERSHIP is the cycle guard ({@link recompleteNode} refuses a node
|
|
314
|
+
* already open on this chain), which is what lets a completion recurse as
|
|
315
|
+
* deep as the graph licenses while the work stays the answer's: the
|
|
316
|
+
* recursion only advances through an ACCEPTED completion, every produced
|
|
317
|
+
* form is decomposed into its own kids, and the memo re-covers each node at
|
|
318
|
+
* most once per cover. A Set, not a flag, because it states WHICH node is
|
|
319
|
+
* open — the invariant a reader needs to check the guard. */
|
|
304
320
|
private recompleteOpen;
|
|
321
|
+
/** JOIN — the move the substrate was missing: derive the answer THROUGH a
|
|
322
|
+
* produced fact, without the intermediate key being named in the query.
|
|
323
|
+
*
|
|
324
|
+
* A produced fact (`fact.node`) carries the subject the query reached but
|
|
325
|
+
* never wrote; the query's remaining tail names the relation to follow from
|
|
326
|
+
* it. The pair IS a learned key — `"<entity><tail>"` — so the rule asks the
|
|
327
|
+
* store for that key's continuation and, when it exists, concludes with the
|
|
328
|
+
* joined fact. On the ladder it is one STEP: a direct edge, exactly as
|
|
329
|
+
* following a literal continuation is. Deterministic and point-probed
|
|
330
|
+
* (`resolve` + `nextFirst`, no scan), so it adds no read that grows with the
|
|
331
|
+
* corpus. The move is visible in the rationale as its own act
|
|
332
|
+
* (`classifyMove` reports `join-fact`), distinct from the byte-concatenating
|
|
333
|
+
* `fuse`/`splice`. */
|
|
334
|
+
private join;
|
|
305
335
|
/** out(i,j,bytes,…): index it for the binary rules, then offer splicing a
|
|
306
336
|
* learnt connector (the in-search bridge), splitting (at a sub-leaf form
|
|
307
|
-
* boundary), bridging (cover(i) ∧ this → cover(j)),
|
|
308
|
-
*
|
|
337
|
+
* boundary), bridging (cover(i) ∧ this → cover(j)), fusing with an adjacent
|
|
338
|
+
* finalised out, and — for a produced fact — JOINING the entity it contains
|
|
339
|
+
* with the query's tail ({@link join}). */
|
|
309
340
|
private outRules;
|
|
310
341
|
/** Whether the query span [from, to) is wholly covered by RECOGNISED outs —
|
|
311
342
|
* the test that lets a connector jump across INTERIOR answers (an N-ary whole)
|
|
@@ -101,8 +101,11 @@ function classifyMove(premises, conclusion, articulating) {
|
|
|
101
101
|
return "step";
|
|
102
102
|
return articulating ? "voice" : "ground";
|
|
103
103
|
}
|
|
104
|
-
if (p.kind === "out" && conclusion.kind === "out")
|
|
105
|
-
|
|
104
|
+
if (p.kind === "out" && conclusion.kind === "out") {
|
|
105
|
+
// A JOIN derives through a produced fact's own contained subject; a plain
|
|
106
|
+
// single-premise out→out is the byte-level split.
|
|
107
|
+
return conclusion.join ? "join-fact" : "split";
|
|
108
|
+
}
|
|
106
109
|
return "step";
|
|
107
110
|
}
|
|
108
111
|
if (premises.length === 2) {
|
|
@@ -230,12 +233,26 @@ export class GraphSearch {
|
|
|
230
233
|
// through (completion is cover, recursively — see {@link recompleteNode}).
|
|
231
234
|
this.recompleteOpen.clear();
|
|
232
235
|
this.recompleteMemo = new Map();
|
|
233
|
-
|
|
236
|
+
// The top cover's derivation sink is threaded into every nested completion
|
|
237
|
+
// so the recompositions a produced form needs are reported in the same
|
|
238
|
+
// trace instead of vanishing after the first layer.
|
|
239
|
+
this.derivationSink = onDerivation;
|
|
240
|
+
const solved = this.solve(queryLen, {
|
|
234
241
|
sites,
|
|
235
242
|
leaves,
|
|
236
243
|
splits,
|
|
237
244
|
starts,
|
|
238
245
|
}, conceptTarget, substitutions, connectors, computedResults, onDerivation);
|
|
246
|
+
// Deepening runs HERE, once, on the derivation the top cover CHOSE — never
|
|
247
|
+
// inside the nested solve a completion runs. Nesting the deepening is what
|
|
248
|
+
// made per-query cost track how densely the corpus interconnects the forms
|
|
249
|
+
// passed through: every re-cover fanned out into the whole corpus's
|
|
250
|
+
// continuations instead of following the chain the answer itself licensed.
|
|
251
|
+
// With deepening only at the top, `recompleteNode` walks the accepted chain
|
|
252
|
+
// one link at a time (its own memo and stack), so the work is the answer's.
|
|
253
|
+
return solved === null
|
|
254
|
+
? null
|
|
255
|
+
: { segs: this.deepen(solved.segs), cost: solved.cost };
|
|
239
256
|
}
|
|
240
257
|
/** Build the deduction system for one span and return its lightest cover's
|
|
241
258
|
* chosen spans — the SINGLE routine the query and every produced composite
|
|
@@ -270,7 +287,7 @@ export class GraphSearch {
|
|
|
270
287
|
onDerivation(readDerivation(derivation, substitutions !== undefined));
|
|
271
288
|
}
|
|
272
289
|
return derivation
|
|
273
|
-
? { segs:
|
|
290
|
+
? { segs: readCover(derivation), cost: derivation.cost }
|
|
274
291
|
: null;
|
|
275
292
|
}
|
|
276
293
|
/** Re-cover the CHOSEN fixpoint spans, in place.
|
|
@@ -311,6 +328,13 @@ export class GraphSearch {
|
|
|
311
328
|
// states the formula itself instead of importing it.
|
|
312
329
|
const atomsAreHubs = Math.max(1, Math.ceil((this.store.edgeSourceCount() * W) / 256)) > this.hubBound();
|
|
313
330
|
const nodeBytes = (n) => this.store.bytesPrefix(n, ALL);
|
|
331
|
+
// The query's own bytes, tiled from its perceived leaves. A JOIN reads the
|
|
332
|
+
// tail a produced fact's contained entity has to combine with, and `buildSearch`
|
|
333
|
+
// otherwise only ever sees positions, never the bytes behind them.
|
|
334
|
+
const queryBytes = new Uint8Array(queryLen);
|
|
335
|
+
for (const lf of leaves) {
|
|
336
|
+
queryBytes.set(lf.bytes.subarray(0, lf.end - lf.start), lf.start);
|
|
337
|
+
}
|
|
314
338
|
// Content-addressed probes over the store's hash-cons maps — the same keys
|
|
315
339
|
// training filled. No byte-by-byte trie walk.
|
|
316
340
|
const findLeafU = (b) => this.store.findLeaf(b) ?? undefined;
|
|
@@ -343,7 +367,7 @@ export class GraphSearch {
|
|
|
343
367
|
if (it.kind === "form") {
|
|
344
368
|
return `f${it.i}.${it.j}.${it.node}.${it.via ? 1 : 0}.${it.rcmp ? 1 : 0}`;
|
|
345
369
|
}
|
|
346
|
-
return `o${it.i}.${it.j}.${it.cover ? 1 : 0}.${it.rec ? 1 : 0}.${it.fix ? 1 : 0}.${it.node ?? -1}.${latin1(it.bytes)}`;
|
|
370
|
+
return `o${it.i}.${it.j}.${it.cover ? 1 : 0}.${it.rec ? 1 : 0}.${it.fix ? 1 : 0}.${it.join ? 1 : 0}.${it.node ?? -1}.${latin1(it.bytes)}`;
|
|
347
371
|
},
|
|
348
372
|
*axioms() {
|
|
349
373
|
yield { item: { kind: "cover", p: 0 }, cost: 0 };
|
|
@@ -451,6 +475,8 @@ export class GraphSearch {
|
|
|
451
475
|
findBranchU,
|
|
452
476
|
linksByLeft,
|
|
453
477
|
linksByRight,
|
|
478
|
+
queryBytes,
|
|
479
|
+
queryLen,
|
|
454
480
|
});
|
|
455
481
|
},
|
|
456
482
|
};
|
|
@@ -722,55 +748,51 @@ export class GraphSearch {
|
|
|
722
748
|
* and recompose into a deeper learnt form (→ FINAL). This is why a single
|
|
723
749
|
* edge-target needs no bespoke logic — it routes back through {@link solve}.
|
|
724
750
|
*
|
|
725
|
-
*
|
|
726
|
-
*
|
|
727
|
-
*
|
|
751
|
+
* The produced bytes are decomposed by the machinery that owns their shape:
|
|
752
|
+
* the span's LEAVES and SPLITS drive the split rule, which resolves each half
|
|
753
|
+
* through findLeaf — so a part that straddles a content-defined cut is still
|
|
754
|
+
* recovered even though the node's tree children need not align with the
|
|
755
|
+
* learnt parts (the fold cuts "p1 p2" as "p1 p"|"2"). Recognised SITES are
|
|
756
|
+
* filtered to the node's own kids: a produced form is completed out of what
|
|
757
|
+
* it was built from, never by re-recognising arbitrary forms inside it. That
|
|
758
|
+
* filter is load-bearing — a 37-byte dialogue sentence carries seven hub
|
|
759
|
+
* openers, and re-covering them chained through the corpus's whole
|
|
760
|
+
* continuation population: 2.5 GB and OOM for `respond("hi.")`.
|
|
728
761
|
*
|
|
729
762
|
* The recovered answer is accepted only when it MOVED and names a LEARNT node
|
|
730
763
|
* ({@link resolve}) — the graph itself gates against re-expanding a contained
|
|
731
|
-
* form ("ice is cold" ⊅→ "ice is cold is cold").
|
|
732
|
-
*
|
|
733
|
-
*
|
|
734
|
-
* another re-cover (see the guard below), so one cover pays at most one
|
|
735
|
-
* nested {@link solve} per distinct produced node it actually reaches, and
|
|
736
|
-
* {@link recompleteMemo} collapses a repeat to nothing.
|
|
764
|
+
* form ("ice is cold" ⊅→ "ice is cold is cold"). An ACCEPTED completion is
|
|
765
|
+
* then re-covered in turn, so a chain runs as deep as the graph licenses; a
|
|
766
|
+
* rejected one ends its branch, so no work is spent past it.
|
|
737
767
|
*
|
|
738
|
-
*
|
|
739
|
-
*
|
|
740
|
-
*
|
|
741
|
-
*
|
|
742
|
-
*
|
|
743
|
-
*
|
|
768
|
+
* Termination and cost are STRUCTURAL: {@link recompleteOpen} is the chain's
|
|
769
|
+
* stack (membership is the cycle guard), {@link recompleteMemo} re-covers each
|
|
770
|
+
* node at most once, only accepted completions recurse, and every level
|
|
771
|
+
* deepens only the top derivation — so a cover pays for the chain it finds,
|
|
772
|
+
* not for how densely the corpus interconnects the forms it passes through.
|
|
773
|
+
* Guard: test/89-completion-recursion.test.mjs. */
|
|
744
774
|
recompleteNode(node) {
|
|
745
775
|
if (!this.host.recogniseSpan)
|
|
746
776
|
return null;
|
|
747
777
|
const memo = this.recompleteMemo;
|
|
748
778
|
if (memo.has(node))
|
|
749
779
|
return memo.get(node) ?? null;
|
|
750
|
-
// ONE re-cover per produced node — never a re-cover inside a re-cover.
|
|
751
|
-
//
|
|
752
780
|
// Re-covering is how a PRODUCED node's bytes enter the search at all: the
|
|
753
|
-
// cover machinery otherwise only ever sees the QUERY's spans.
|
|
754
|
-
//
|
|
755
|
-
//
|
|
756
|
-
//
|
|
757
|
-
//
|
|
758
|
-
//
|
|
759
|
-
// bytes nothing else brings in) needs it.
|
|
760
|
-
//
|
|
761
|
-
// Nesting it was the defect. Each level is a full {@link solve} with its
|
|
762
|
-
// own agenda and chart, exploring from a node the answer never asked about,
|
|
763
|
-
// so per-query cost tracked how densely the corpus interconnects the forms
|
|
764
|
-
// passed through — the growth AGENTS §2.8 forbids. Measured on an
|
|
765
|
-
// 18.9M-node store: depth 331 and 9.1 GB for a 2-byte query, not
|
|
766
|
-
// terminating; and on the guard corpus every one of 125 nested re-covers
|
|
767
|
-
// was REJECTED by the resolve() gate below, expanding a 70-byte node into a
|
|
768
|
-
// 374-byte concatenation that names nothing. All of it was waste.
|
|
781
|
+
// cover machinery otherwise only ever sees the QUERY's spans. The recursion
|
|
782
|
+
// is allowed to nest — a chain IS nested completions — but it is bounded so
|
|
783
|
+
// the work stays the ANSWER's (AGENTS §2.8): the stack below is the cycle
|
|
784
|
+
// guard, only ACCEPTED completions recurse, and the nested solve decomposes
|
|
785
|
+
// the form by its own shape instead of re-recognising the corpus's hub forms
|
|
786
|
+
// inside it.
|
|
769
787
|
//
|
|
770
|
-
// `recompleteOpen`
|
|
771
|
-
//
|
|
772
|
-
//
|
|
773
|
-
|
|
788
|
+
// `recompleteOpen` IS the stack of the chain being built, so MEMBERSHIP is
|
|
789
|
+
// the cycle guard: a node already open on this chain cannot re-enter it.
|
|
790
|
+
// Testing the NODE — not the stack's size — is what lets a completion
|
|
791
|
+
// recurse as deep as the graph licenses, exactly the intrinsic convergence
|
|
792
|
+
// {@link solve}'s contract states. Work stays the answer's because the
|
|
793
|
+
// recursion only ever advances through an ACCEPTED completion (below) and
|
|
794
|
+
// {@link cover} deepens only the top derivation.
|
|
795
|
+
if (this.recompleteOpen.has(node))
|
|
774
796
|
return null;
|
|
775
797
|
// A leaf or single-child node has no parts to recompose; skip before the
|
|
776
798
|
// costly recognition so a plain terminal answer pays nothing.
|
|
@@ -782,16 +804,43 @@ export class GraphSearch {
|
|
|
782
804
|
const bytes = this.store.bytesPrefix(node, ALL);
|
|
783
805
|
this.recompleteOpen.add(node);
|
|
784
806
|
try {
|
|
785
|
-
// Completion is cover
|
|
786
|
-
//
|
|
787
|
-
//
|
|
788
|
-
//
|
|
789
|
-
|
|
807
|
+
// Completion is cover, but the produced form is decomposed by its own
|
|
808
|
+
// shape. The LEAVES and SPLITS are kept whole, so the split rule still
|
|
809
|
+
// recovers a part that straddles a content-defined cut (the fold cuts
|
|
810
|
+
// "p1 p2" as "p1 p"|"2", and findLeaf still resolves p1 and p2). The
|
|
811
|
+
// recognised SITES are filtered to the node's own kids, because
|
|
812
|
+
// re-recognising arbitrary forms inside the bytes is what let a hub-heavy
|
|
813
|
+
// utterance explode: a 37-byte dialogue sentence carries seven hub
|
|
814
|
+
// openers, and re-covering them chained through the corpus's whole
|
|
815
|
+
// continuation population (2.5 GB and OOM for `"hi."`). No
|
|
816
|
+
// concepts/connectors either (those need the caller's async
|
|
817
|
+
// pre-resolution) — the recursion follows edges and fusion, which is what
|
|
818
|
+
// a deeper rewrite chain is made of.
|
|
819
|
+
const rec = this.host.recogniseSpan(bytes);
|
|
820
|
+
const kids = new Set(nrec.kids);
|
|
821
|
+
const solved = this.solve(bytes.length, {
|
|
822
|
+
sites: rec.sites.filter((s) => kids.has(s.payload)),
|
|
823
|
+
leaves: rec.leaves,
|
|
824
|
+
splits: rec.splits,
|
|
825
|
+
starts: rec.starts,
|
|
826
|
+
}, new Map(), undefined, undefined, undefined, this.derivationSink);
|
|
790
827
|
const answer = solved && concatBytes(solved.segs.map((s) => s.bytes));
|
|
791
|
-
|
|
792
|
-
|
|
793
|
-
|
|
828
|
+
// ACCEPT, then CONTINUE THE CHAIN — but only along an accepted
|
|
829
|
+
// completion. A re-cover whose result is not itself a learnt node (the
|
|
830
|
+
// 70→374-byte concatenations that name nothing) ends its branch here
|
|
831
|
+
// instead of recursing into work the answer never asked for, and an
|
|
832
|
+
// accepted one names a node that is re-covered in turn. That is what
|
|
833
|
+
// keeps a deep chain possible while the work stays proportional to the
|
|
834
|
+
// chain rather than to the corpus's interconnections.
|
|
835
|
+
const composed = answer !== null && !bytesEqual(answer, bytes)
|
|
836
|
+
? this.host.resolve(answer)
|
|
794
837
|
: null;
|
|
838
|
+
if (composed === null) {
|
|
839
|
+
memo.set(node, null);
|
|
840
|
+
return null;
|
|
841
|
+
}
|
|
842
|
+
const deeper = this.recompleteNode(composed);
|
|
843
|
+
const out = deeper ?? answer;
|
|
795
844
|
memo.set(node, out);
|
|
796
845
|
return out;
|
|
797
846
|
}
|
|
@@ -803,18 +852,76 @@ export class GraphSearch {
|
|
|
803
852
|
* outs of a long query re-cover each distinct node at most once); reset at the
|
|
804
853
|
* top of {@link cover}. */
|
|
805
854
|
recompleteMemo = new Map();
|
|
806
|
-
/** The
|
|
807
|
-
*
|
|
808
|
-
*
|
|
809
|
-
* is
|
|
810
|
-
|
|
811
|
-
|
|
812
|
-
*
|
|
855
|
+
/** The derivation sink of the TOP cover, threaded into every nested
|
|
856
|
+
* completion so a produced form's own recompositions are reported in the
|
|
857
|
+
* same trace instead of vanishing after the first layer. Undefined when
|
|
858
|
+
* nothing is inspecting, so an uninspected response pays nothing. */
|
|
859
|
+
derivationSink;
|
|
860
|
+
/** The chain of nodes currently being re-completed — the recursion STACK.
|
|
861
|
+
* MEMBERSHIP is the cycle guard ({@link recompleteNode} refuses a node
|
|
862
|
+
* already open on this chain), which is what lets a completion recurse as
|
|
863
|
+
* deep as the graph licenses while the work stays the answer's: the
|
|
864
|
+
* recursion only advances through an ACCEPTED completion, every produced
|
|
865
|
+
* form is decomposed into its own kids, and the memo re-covers each node at
|
|
866
|
+
* most once per cover. A Set, not a flag, because it states WHICH node is
|
|
867
|
+
* open — the invariant a reader needs to check the guard. */
|
|
813
868
|
recompleteOpen = new Set();
|
|
869
|
+
/** JOIN — the move the substrate was missing: derive the answer THROUGH a
|
|
870
|
+
* produced fact, without the intermediate key being named in the query.
|
|
871
|
+
*
|
|
872
|
+
* A produced fact (`fact.node`) carries the subject the query reached but
|
|
873
|
+
* never wrote; the query's remaining tail names the relation to follow from
|
|
874
|
+
* it. The pair IS a learned key — `"<entity><tail>"` — so the rule asks the
|
|
875
|
+
* store for that key's continuation and, when it exists, concludes with the
|
|
876
|
+
* joined fact. On the ladder it is one STEP: a direct edge, exactly as
|
|
877
|
+
* following a literal continuation is. Deterministic and point-probed
|
|
878
|
+
* (`resolve` + `nextFirst`, no scan), so it adds no read that grows with the
|
|
879
|
+
* corpus. The move is visible in the rationale as its own act
|
|
880
|
+
* (`classifyMove` reports `join-fact`), distinct from the byte-concatenating
|
|
881
|
+
* `fuse`/`splice`. */
|
|
882
|
+
*join(fact, queryBytes, queryLen) {
|
|
883
|
+
if (!this.host.recogniseSpan)
|
|
884
|
+
return;
|
|
885
|
+
const tail = queryBytes.subarray(fact.j, queryLen);
|
|
886
|
+
if (tail.length === 0)
|
|
887
|
+
return;
|
|
888
|
+
// The entity candidates are the forms the fact's own bytes CONTAIN — the
|
|
889
|
+
// same recogniser the query went through, so the evidence standard is the
|
|
890
|
+
// query's. A byte atom is never a subject; the fact's own node is the span
|
|
891
|
+
// itself, not an entity inside it.
|
|
892
|
+
for (const site of this.host.recogniseSpan(fact.bytes).sites) {
|
|
893
|
+
if (site.payload < 0 || site.payload === fact.node)
|
|
894
|
+
continue;
|
|
895
|
+
if (!this.store.hasNext(site.payload) &&
|
|
896
|
+
!this.store.hasHalo(site.payload))
|
|
897
|
+
continue;
|
|
898
|
+
const key = this.host.resolve(concat2(this.store.bytesPrefix(site.payload, ALL), tail));
|
|
899
|
+
if (key === null)
|
|
900
|
+
continue;
|
|
901
|
+
const nx = this.store.nextFirst(key, 1);
|
|
902
|
+
if (nx.length === 0)
|
|
903
|
+
continue;
|
|
904
|
+
yield {
|
|
905
|
+
premises: [fact],
|
|
906
|
+
conclusion: {
|
|
907
|
+
kind: "out",
|
|
908
|
+
i: fact.i,
|
|
909
|
+
j: queryLen,
|
|
910
|
+
bytes: this.store.bytesPrefix(nx[0], ALL),
|
|
911
|
+
cover: true,
|
|
912
|
+
rec: true,
|
|
913
|
+
node: nx[0],
|
|
914
|
+
join: true,
|
|
915
|
+
},
|
|
916
|
+
cost: STEP,
|
|
917
|
+
};
|
|
918
|
+
}
|
|
919
|
+
}
|
|
814
920
|
/** out(i,j,bytes,…): index it for the binary rules, then offer splicing a
|
|
815
921
|
* learnt connector (the in-search bridge), splitting (at a sub-leaf form
|
|
816
|
-
* boundary), bridging (cover(i) ∧ this → cover(j)),
|
|
817
|
-
*
|
|
922
|
+
* boundary), bridging (cover(i) ∧ this → cover(j)), fusing with an adjacent
|
|
923
|
+
* finalised out, and — for a produced fact — JOINING the entity it contains
|
|
924
|
+
* with the query's tail ({@link join}). */
|
|
818
925
|
*outRules(it, ctx) {
|
|
819
926
|
const { splits, coversDone, outsByStart, outsByEnd, coverableByStart } = ctx;
|
|
820
927
|
const outsByNode = ctx.outsByNode;
|
|
@@ -907,6 +1014,17 @@ export class GraphSearch {
|
|
|
907
1014
|
yield* this.fuse(it, r, ctx);
|
|
908
1015
|
for (const l of outsByEnd.get(it.i) ?? [])
|
|
909
1016
|
yield* this.fuse(l, it, ctx);
|
|
1017
|
+
// ── JOIN (the A*LD extension) ───────────────────────────────────────
|
|
1018
|
+
// A produced fact may CONTAIN the subject the query never named; the query's
|
|
1019
|
+
// remaining tail then names the relation to follow FROM that subject. The
|
|
1020
|
+
// pair (contained entity, tail) is itself a learned key, and its
|
|
1021
|
+
// continuation is the derived answer — a genuine two-fact join, not the
|
|
1022
|
+
// juxtaposition the cover produces when the intermediate key IS named.
|
|
1023
|
+
// Fired per finalized out with a node, so it is the search's own rule, on
|
|
1024
|
+
// the ladder, memoised by {@link key}, and bounded by the fact's own length.
|
|
1025
|
+
if (it.node !== undefined) {
|
|
1026
|
+
yield* this.join(it, ctx.queryBytes, ctx.queryLen);
|
|
1027
|
+
}
|
|
910
1028
|
}
|
|
911
1029
|
/** Whether the query span [from, to) is wholly covered by RECOGNISED outs —
|
|
912
1030
|
* the test that lets a connector jump across INTERIOR answers (an N-ary whole)
|
|
@@ -501,7 +501,21 @@ function recogniseImpl(ctx, bytes) {
|
|
|
501
501
|
// embedded differently-cased form needed 64x the budget to be found,
|
|
502
502
|
// while the exact route it was competing with needed none of it.
|
|
503
503
|
const probe = (start, end, canonBudget) => {
|
|
504
|
-
|
|
504
|
+
// Any span at least one river window wide is worth a probe. This used
|
|
505
|
+
// to stop at `chainReach(W)` — "the chain already covers anything that
|
|
506
|
+
// short" — and that premise is false for a NESTED form: the chain grows
|
|
507
|
+
// single-byte leaf ids and gates each step on `findBranch(ids)`, which
|
|
508
|
+
// is null for a form the write side chunked (measured: "Gustaf
|
|
509
|
+
// Molander" embedded in "The director of Eva is Gustaf Molander." gains
|
|
510
|
+
// no branch at any prefix, so `resolveSpan` is never reached), and its
|
|
511
|
+
// INTERIOR reach is one chunk plus W, which can be shorter than the
|
|
512
|
+
// form. The result was a dead zone: a form shorter than `chainReach`
|
|
513
|
+
// that neither starts on a fold cut nor ends on a node edge was
|
|
514
|
+
// unreachable by either tier — the exact site whose loss `tryChain`'s
|
|
515
|
+
// own note records as "the pivot dies with the site and multi-hop goes
|
|
516
|
+
// silent". The interior pass below spends the same budget on those
|
|
517
|
+
// pairs.
|
|
518
|
+
if (end - start < W)
|
|
505
519
|
return;
|
|
506
520
|
if (flatProbe(start, end) === null) {
|
|
507
521
|
if (!canonBudget)
|
|
@@ -565,6 +579,26 @@ function recogniseImpl(ctx, bytes) {
|
|
|
565
579
|
if (i < suffixes.length && !spend(suffixes[i], bytes.length))
|
|
566
580
|
break;
|
|
567
581
|
}
|
|
582
|
+
// INTERIOR pairs, bounded by the same `chainReach(W)` the chain trusts —
|
|
583
|
+
// the dead zone the gate above used to leave: a form that neither starts
|
|
584
|
+
// on a fold cut nor ends on a node edge is exactly the one neither the
|
|
585
|
+
// chain (nested, `findBranch` misses) nor the two edge scans reach. The
|
|
586
|
+
// pair count is `|endpoints| · chainReach(W)`, i.e. LINEAR in the query —
|
|
587
|
+
// the W² span bound is what keeps this from being the quadratic scan the
|
|
588
|
+
// budget note above describes (that one had no span bound at all).
|
|
589
|
+
{
|
|
590
|
+
const reach = chainReach(W);
|
|
591
|
+
for (const end of ordered) {
|
|
592
|
+
for (const start of ordered) {
|
|
593
|
+
if (start >= end)
|
|
594
|
+
continue;
|
|
595
|
+
const span = end - start;
|
|
596
|
+
if (span < W || span > reach)
|
|
597
|
+
continue;
|
|
598
|
+
spend(start, end);
|
|
599
|
+
}
|
|
600
|
+
}
|
|
601
|
+
}
|
|
568
602
|
}
|
|
569
603
|
}
|
|
570
604
|
const chunkEnd = new Uint32Array(bytes.length);
|
package/dist/src/mind/trace.js
CHANGED
|
@@ -50,6 +50,7 @@ export const MOVE_NOTE = {
|
|
|
50
50
|
"split": "cut a span at a sub-leaf form boundary so a form can be reached",
|
|
51
51
|
"fuse": "fuse adjacent fragments toward a deeper learned form",
|
|
52
52
|
"recompose": "recompose fused parts into a learned whole that leads on",
|
|
53
|
+
"join-fact": "join a produced fact's own subject with the query's tail — derive through the fact, not alongside it",
|
|
53
54
|
"bridge": "advance the cover frontier across this span",
|
|
54
55
|
"pool-vote": "pool independent regions' evidence for a shared anchor (sum, not shortest path)",
|
|
55
56
|
"axiom": "a seed: a perceived leaf, recognised form, or computed result",
|
package/jsr.json
CHANGED
package/package.json
CHANGED
package/src/mind/graph-search.ts
CHANGED
|
@@ -115,6 +115,12 @@ export type GItem =
|
|
|
115
115
|
* derivation actually CHOSE. Part of {@link key}, because it decides
|
|
116
116
|
* whether the span's final bytes may still change. */
|
|
117
117
|
fix?: boolean;
|
|
118
|
+
/** Set on the out a JOIN produced: a produced fact's own contained entity
|
|
119
|
+
* (the subject the query never named) combined with the query's adjacent
|
|
120
|
+
* relation span named a learned key, and that key's continuation is this
|
|
121
|
+
* out. Part of {@link key} so the joined reading is a distinct chart item
|
|
122
|
+
* from the plain concatenation of the same bytes. */
|
|
123
|
+
join?: boolean;
|
|
118
124
|
};
|
|
119
125
|
type OutItem = Extract<GItem, { kind: "out" }>;
|
|
120
126
|
|
|
@@ -240,6 +246,7 @@ export type DerivationMove =
|
|
|
240
246
|
| "split" // out→out cut at a sub-leaf form boundary
|
|
241
247
|
| "fuse" // out+out→out: adjacent fragments recomposed toward a learned form
|
|
242
248
|
| "recompose" // out+out→form: a fused pair that names an edge-bearing node
|
|
249
|
+
| "join-fact" // out→out: a produced fact's own contained subject + the query's tail names a learned key (the join)
|
|
243
250
|
| "bridge" // cover+out→cover: the cover frontier advanced across a span
|
|
244
251
|
| "pool-vote" // N premises→conclusion, evidence pooled (combine:"sum" — see derive)
|
|
245
252
|
| "step"; // any other single-premise move (fallback)
|
|
@@ -276,7 +283,11 @@ function classifyMove(
|
|
|
276
283
|
if (!conclusion.rec) return "step";
|
|
277
284
|
return articulating ? "voice" : "ground";
|
|
278
285
|
}
|
|
279
|
-
if (p.kind === "out" && conclusion.kind === "out")
|
|
286
|
+
if (p.kind === "out" && conclusion.kind === "out") {
|
|
287
|
+
// A JOIN derives through a produced fact's own contained subject; a plain
|
|
288
|
+
// single-premise out→out is the byte-level split.
|
|
289
|
+
return conclusion.join ? "join-fact" : "split";
|
|
290
|
+
}
|
|
280
291
|
return "step";
|
|
281
292
|
}
|
|
282
293
|
if (premises.length === 2) {
|
|
@@ -412,7 +423,11 @@ export class GraphSearch {
|
|
|
412
423
|
// through (completion is cover, recursively — see {@link recompleteNode}).
|
|
413
424
|
this.recompleteOpen.clear();
|
|
414
425
|
this.recompleteMemo = new Map<number, Uint8Array | null>();
|
|
415
|
-
|
|
426
|
+
// The top cover's derivation sink is threaded into every nested completion
|
|
427
|
+
// so the recompositions a produced form needs are reported in the same
|
|
428
|
+
// trace instead of vanishing after the first layer.
|
|
429
|
+
this.derivationSink = onDerivation;
|
|
430
|
+
const solved = this.solve(
|
|
416
431
|
queryLen,
|
|
417
432
|
{
|
|
418
433
|
sites,
|
|
@@ -426,8 +441,17 @@ export class GraphSearch {
|
|
|
426
441
|
computedResults,
|
|
427
442
|
onDerivation,
|
|
428
443
|
);
|
|
444
|
+
// Deepening runs HERE, once, on the derivation the top cover CHOSE — never
|
|
445
|
+
// inside the nested solve a completion runs. Nesting the deepening is what
|
|
446
|
+
// made per-query cost track how densely the corpus interconnects the forms
|
|
447
|
+
// passed through: every re-cover fanned out into the whole corpus's
|
|
448
|
+
// continuations instead of following the chain the answer itself licensed.
|
|
449
|
+
// With deepening only at the top, `recompleteNode` walks the accepted chain
|
|
450
|
+
// one link at a time (its own memo and stack), so the work is the answer's.
|
|
451
|
+
return solved === null
|
|
452
|
+
? null
|
|
453
|
+
: { segs: this.deepen(solved.segs), cost: solved.cost };
|
|
429
454
|
}
|
|
430
|
-
|
|
431
455
|
/** Build the deduction system for one span and return its lightest cover's
|
|
432
456
|
* chosen spans — the SINGLE routine the query and every produced composite
|
|
433
457
|
* run through. `recognition` carries the span's recognised forms; the query
|
|
@@ -484,7 +508,7 @@ export class GraphSearch {
|
|
|
484
508
|
onDerivation(readDerivation(derivation, substitutions !== undefined));
|
|
485
509
|
}
|
|
486
510
|
return derivation
|
|
487
|
-
? { segs:
|
|
511
|
+
? { segs: readCover(derivation), cost: derivation.cost }
|
|
488
512
|
: null;
|
|
489
513
|
}
|
|
490
514
|
|
|
@@ -538,6 +562,13 @@ export class GraphSearch {
|
|
|
538
562
|
Math.ceil((this.store.edgeSourceCount() * W) / 256),
|
|
539
563
|
) > this.hubBound();
|
|
540
564
|
const nodeBytes = (n: number) => this.store.bytesPrefix(n, ALL);
|
|
565
|
+
// The query's own bytes, tiled from its perceived leaves. A JOIN reads the
|
|
566
|
+
// tail a produced fact's contained entity has to combine with, and `buildSearch`
|
|
567
|
+
// otherwise only ever sees positions, never the bytes behind them.
|
|
568
|
+
const queryBytes = new Uint8Array(queryLen);
|
|
569
|
+
for (const lf of leaves) {
|
|
570
|
+
queryBytes.set(lf.bytes.subarray(0, lf.end - lf.start), lf.start);
|
|
571
|
+
}
|
|
541
572
|
// Content-addressed probes over the store's hash-cons maps — the same keys
|
|
542
573
|
// training filled. No byte-by-byte trie walk.
|
|
543
574
|
const findLeafU = (b: Uint8Array) => this.store.findLeaf(b) ?? undefined;
|
|
@@ -576,7 +607,7 @@ export class GraphSearch {
|
|
|
576
607
|
}
|
|
577
608
|
return `o${it.i}.${it.j}.${it.cover ? 1 : 0}.${it.rec ? 1 : 0}.${
|
|
578
609
|
it.fix ? 1 : 0
|
|
579
|
-
}.${it.node ?? -1}.${latin1(it.bytes)}`;
|
|
610
|
+
}.${it.join ? 1 : 0}.${it.node ?? -1}.${latin1(it.bytes)}`;
|
|
580
611
|
},
|
|
581
612
|
*axioms() {
|
|
582
613
|
yield { item: { kind: "cover", p: 0 }, cost: 0 };
|
|
@@ -684,6 +715,8 @@ export class GraphSearch {
|
|
|
684
715
|
findBranchU,
|
|
685
716
|
linksByLeft,
|
|
686
717
|
linksByRight,
|
|
718
|
+
queryBytes,
|
|
719
|
+
queryLen,
|
|
687
720
|
});
|
|
688
721
|
},
|
|
689
722
|
};
|
|
@@ -974,53 +1007,49 @@ export class GraphSearch {
|
|
|
974
1007
|
* and recompose into a deeper learnt form (→ FINAL). This is why a single
|
|
975
1008
|
* edge-target needs no bespoke logic — it routes back through {@link solve}.
|
|
976
1009
|
*
|
|
977
|
-
*
|
|
978
|
-
*
|
|
979
|
-
*
|
|
1010
|
+
* The produced bytes are decomposed by the machinery that owns their shape:
|
|
1011
|
+
* the span's LEAVES and SPLITS drive the split rule, which resolves each half
|
|
1012
|
+
* through findLeaf — so a part that straddles a content-defined cut is still
|
|
1013
|
+
* recovered even though the node's tree children need not align with the
|
|
1014
|
+
* learnt parts (the fold cuts "p1 p2" as "p1 p"|"2"). Recognised SITES are
|
|
1015
|
+
* filtered to the node's own kids: a produced form is completed out of what
|
|
1016
|
+
* it was built from, never by re-recognising arbitrary forms inside it. That
|
|
1017
|
+
* filter is load-bearing — a 37-byte dialogue sentence carries seven hub
|
|
1018
|
+
* openers, and re-covering them chained through the corpus's whole
|
|
1019
|
+
* continuation population: 2.5 GB and OOM for `respond("hi.")`.
|
|
980
1020
|
*
|
|
981
1021
|
* The recovered answer is accepted only when it MOVED and names a LEARNT node
|
|
982
1022
|
* ({@link resolve}) — the graph itself gates against re-expanding a contained
|
|
983
|
-
* form ("ice is cold" ⊅→ "ice is cold is cold").
|
|
1023
|
+
* form ("ice is cold" ⊅→ "ice is cold is cold"). An ACCEPTED completion is
|
|
1024
|
+
* then re-covered in turn, so a chain runs as deep as the graph licenses; a
|
|
1025
|
+
* rejected one ends its branch, so no work is spent past it.
|
|
984
1026
|
*
|
|
985
|
-
* Termination
|
|
986
|
-
*
|
|
987
|
-
*
|
|
988
|
-
*
|
|
989
|
-
*
|
|
990
|
-
*
|
|
991
|
-
* and each finished completion is memoised". That is a bound of N — the one
|
|
992
|
-
* AGENTS §2.8 forbids — and it was load-bearing, not pedantic: nested, the
|
|
993
|
-
* recursion reached depth 331 and 9.1 GB on an 18.9M-node store for a 2-byte
|
|
994
|
-
* query and did not terminate, which is what killed a 5 h training run at its
|
|
995
|
-
* checkpoint recall. Guard: test/89-completion-recursion.test.mjs. */
|
|
1027
|
+
* Termination and cost are STRUCTURAL: {@link recompleteOpen} is the chain's
|
|
1028
|
+
* stack (membership is the cycle guard), {@link recompleteMemo} re-covers each
|
|
1029
|
+
* node at most once, only accepted completions recurse, and every level
|
|
1030
|
+
* deepens only the top derivation — so a cover pays for the chain it finds,
|
|
1031
|
+
* not for how densely the corpus interconnects the forms it passes through.
|
|
1032
|
+
* Guard: test/89-completion-recursion.test.mjs. */
|
|
996
1033
|
private recompleteNode(node: number): Uint8Array | null {
|
|
997
1034
|
if (!this.host.recogniseSpan) return null;
|
|
998
1035
|
const memo = this.recompleteMemo;
|
|
999
1036
|
if (memo.has(node)) return memo.get(node) ?? null;
|
|
1000
|
-
// ONE re-cover per produced node — never a re-cover inside a re-cover.
|
|
1001
|
-
//
|
|
1002
1037
|
// Re-covering is how a PRODUCED node's bytes enter the search at all: the
|
|
1003
|
-
// cover machinery otherwise only ever sees the QUERY's spans.
|
|
1004
|
-
//
|
|
1005
|
-
//
|
|
1006
|
-
//
|
|
1007
|
-
//
|
|
1008
|
-
//
|
|
1009
|
-
// bytes nothing else brings in) needs it.
|
|
1010
|
-
//
|
|
1011
|
-
// Nesting it was the defect. Each level is a full {@link solve} with its
|
|
1012
|
-
// own agenda and chart, exploring from a node the answer never asked about,
|
|
1013
|
-
// so per-query cost tracked how densely the corpus interconnects the forms
|
|
1014
|
-
// passed through — the growth AGENTS §2.8 forbids. Measured on an
|
|
1015
|
-
// 18.9M-node store: depth 331 and 9.1 GB for a 2-byte query, not
|
|
1016
|
-
// terminating; and on the guard corpus every one of 125 nested re-covers
|
|
1017
|
-
// was REJECTED by the resolve() gate below, expanding a 70-byte node into a
|
|
1018
|
-
// 374-byte concatenation that names nothing. All of it was waste.
|
|
1038
|
+
// cover machinery otherwise only ever sees the QUERY's spans. The recursion
|
|
1039
|
+
// is allowed to nest — a chain IS nested completions — but it is bounded so
|
|
1040
|
+
// the work stays the ANSWER's (AGENTS §2.8): the stack below is the cycle
|
|
1041
|
+
// guard, only ACCEPTED completions recurse, and the nested solve decomposes
|
|
1042
|
+
// the form by its own shape instead of re-recognising the corpus's hub forms
|
|
1043
|
+
// inside it.
|
|
1019
1044
|
//
|
|
1020
|
-
// `recompleteOpen`
|
|
1021
|
-
//
|
|
1022
|
-
//
|
|
1023
|
-
|
|
1045
|
+
// `recompleteOpen` IS the stack of the chain being built, so MEMBERSHIP is
|
|
1046
|
+
// the cycle guard: a node already open on this chain cannot re-enter it.
|
|
1047
|
+
// Testing the NODE — not the stack's size — is what lets a completion
|
|
1048
|
+
// recurse as deep as the graph licenses, exactly the intrinsic convergence
|
|
1049
|
+
// {@link solve}'s contract states. Work stays the answer's because the
|
|
1050
|
+
// recursion only ever advances through an ACCEPTED completion (below) and
|
|
1051
|
+
// {@link cover} deepens only the top derivation.
|
|
1052
|
+
if (this.recompleteOpen.has(node)) return null;
|
|
1024
1053
|
|
|
1025
1054
|
// A leaf or single-child node has no parts to recompose; skip before the
|
|
1026
1055
|
// costly recognition so a plain terminal answer pays nothing.
|
|
@@ -1033,20 +1062,51 @@ export class GraphSearch {
|
|
|
1033
1062
|
const bytes = this.store.bytesPrefix(node, ALL);
|
|
1034
1063
|
this.recompleteOpen.add(node);
|
|
1035
1064
|
try {
|
|
1036
|
-
// Completion is cover
|
|
1037
|
-
//
|
|
1038
|
-
//
|
|
1039
|
-
//
|
|
1065
|
+
// Completion is cover, but the produced form is decomposed by its own
|
|
1066
|
+
// shape. The LEAVES and SPLITS are kept whole, so the split rule still
|
|
1067
|
+
// recovers a part that straddles a content-defined cut (the fold cuts
|
|
1068
|
+
// "p1 p2" as "p1 p"|"2", and findLeaf still resolves p1 and p2). The
|
|
1069
|
+
// recognised SITES are filtered to the node's own kids, because
|
|
1070
|
+
// re-recognising arbitrary forms inside the bytes is what let a hub-heavy
|
|
1071
|
+
// utterance explode: a 37-byte dialogue sentence carries seven hub
|
|
1072
|
+
// openers, and re-covering them chained through the corpus's whole
|
|
1073
|
+
// continuation population (2.5 GB and OOM for `"hi."`). No
|
|
1074
|
+
// concepts/connectors either (those need the caller's async
|
|
1075
|
+
// pre-resolution) — the recursion follows edges and fusion, which is what
|
|
1076
|
+
// a deeper rewrite chain is made of.
|
|
1077
|
+
const rec = this.host.recogniseSpan(bytes);
|
|
1078
|
+
const kids = new Set(nrec.kids);
|
|
1040
1079
|
const solved = this.solve(
|
|
1041
1080
|
bytes.length,
|
|
1042
|
-
|
|
1081
|
+
{
|
|
1082
|
+
sites: rec.sites.filter((s) => kids.has(s.payload)),
|
|
1083
|
+
leaves: rec.leaves,
|
|
1084
|
+
splits: rec.splits,
|
|
1085
|
+
starts: rec.starts,
|
|
1086
|
+
},
|
|
1043
1087
|
new Map(),
|
|
1088
|
+
undefined,
|
|
1089
|
+
undefined,
|
|
1090
|
+
undefined,
|
|
1091
|
+
this.derivationSink,
|
|
1044
1092
|
);
|
|
1045
1093
|
const answer = solved && concatBytes(solved.segs.map((s) => s.bytes));
|
|
1046
|
-
|
|
1047
|
-
|
|
1048
|
-
|
|
1094
|
+
// ACCEPT, then CONTINUE THE CHAIN — but only along an accepted
|
|
1095
|
+
// completion. A re-cover whose result is not itself a learnt node (the
|
|
1096
|
+
// 70→374-byte concatenations that name nothing) ends its branch here
|
|
1097
|
+
// instead of recursing into work the answer never asked for, and an
|
|
1098
|
+
// accepted one names a node that is re-covered in turn. That is what
|
|
1099
|
+
// keeps a deep chain possible while the work stays proportional to the
|
|
1100
|
+
// chain rather than to the corpus's interconnections.
|
|
1101
|
+
const composed = answer !== null && !bytesEqual(answer, bytes)
|
|
1102
|
+
? this.host.resolve(answer)
|
|
1049
1103
|
: null;
|
|
1104
|
+
if (composed === null) {
|
|
1105
|
+
memo.set(node, null);
|
|
1106
|
+
return null;
|
|
1107
|
+
}
|
|
1108
|
+
const deeper = this.recompleteNode(composed);
|
|
1109
|
+
const out = deeper ?? answer!;
|
|
1050
1110
|
memo.set(node, out);
|
|
1051
1111
|
return out;
|
|
1052
1112
|
} finally {
|
|
@@ -1058,19 +1118,80 @@ export class GraphSearch {
|
|
|
1058
1118
|
* outs of a long query re-cover each distinct node at most once); reset at the
|
|
1059
1119
|
* top of {@link cover}. */
|
|
1060
1120
|
private recompleteMemo = new Map<number, Uint8Array | null>();
|
|
1061
|
-
/** The
|
|
1062
|
-
*
|
|
1063
|
-
*
|
|
1064
|
-
* is
|
|
1065
|
-
|
|
1066
|
-
|
|
1067
|
-
*
|
|
1121
|
+
/** The derivation sink of the TOP cover, threaded into every nested
|
|
1122
|
+
* completion so a produced form's own recompositions are reported in the
|
|
1123
|
+
* same trace instead of vanishing after the first layer. Undefined when
|
|
1124
|
+
* nothing is inspecting, so an uninspected response pays nothing. */
|
|
1125
|
+
private derivationSink?: (steps: DerivationStep[]) => void;
|
|
1126
|
+
/** The chain of nodes currently being re-completed — the recursion STACK.
|
|
1127
|
+
* MEMBERSHIP is the cycle guard ({@link recompleteNode} refuses a node
|
|
1128
|
+
* already open on this chain), which is what lets a completion recurse as
|
|
1129
|
+
* deep as the graph licenses while the work stays the answer's: the
|
|
1130
|
+
* recursion only advances through an ACCEPTED completion, every produced
|
|
1131
|
+
* form is decomposed into its own kids, and the memo re-covers each node at
|
|
1132
|
+
* most once per cover. A Set, not a flag, because it states WHICH node is
|
|
1133
|
+
* open — the invariant a reader needs to check the guard. */
|
|
1068
1134
|
private recompleteOpen = new Set<number>();
|
|
1069
1135
|
|
|
1136
|
+
/** JOIN — the move the substrate was missing: derive the answer THROUGH a
|
|
1137
|
+
* produced fact, without the intermediate key being named in the query.
|
|
1138
|
+
*
|
|
1139
|
+
* A produced fact (`fact.node`) carries the subject the query reached but
|
|
1140
|
+
* never wrote; the query's remaining tail names the relation to follow from
|
|
1141
|
+
* it. The pair IS a learned key — `"<entity><tail>"` — so the rule asks the
|
|
1142
|
+
* store for that key's continuation and, when it exists, concludes with the
|
|
1143
|
+
* joined fact. On the ladder it is one STEP: a direct edge, exactly as
|
|
1144
|
+
* following a literal continuation is. Deterministic and point-probed
|
|
1145
|
+
* (`resolve` + `nextFirst`, no scan), so it adds no read that grows with the
|
|
1146
|
+
* corpus. The move is visible in the rationale as its own act
|
|
1147
|
+
* (`classifyMove` reports `join-fact`), distinct from the byte-concatenating
|
|
1148
|
+
* `fuse`/`splice`. */
|
|
1149
|
+
private *join(
|
|
1150
|
+
fact: OutItem,
|
|
1151
|
+
queryBytes: Uint8Array,
|
|
1152
|
+
queryLen: number,
|
|
1153
|
+
): Iterable<Rule<GItem>> {
|
|
1154
|
+
if (!this.host.recogniseSpan) return;
|
|
1155
|
+
const tail = queryBytes.subarray(fact.j, queryLen);
|
|
1156
|
+
if (tail.length === 0) return;
|
|
1157
|
+
// The entity candidates are the forms the fact's own bytes CONTAIN — the
|
|
1158
|
+
// same recogniser the query went through, so the evidence standard is the
|
|
1159
|
+
// query's. A byte atom is never a subject; the fact's own node is the span
|
|
1160
|
+
// itself, not an entity inside it.
|
|
1161
|
+
for (const site of this.host.recogniseSpan(fact.bytes).sites) {
|
|
1162
|
+
if (site.payload < 0 || site.payload === fact.node) continue;
|
|
1163
|
+
if (
|
|
1164
|
+
!this.store.hasNext(site.payload) &&
|
|
1165
|
+
!this.store.hasHalo(site.payload)
|
|
1166
|
+
) continue;
|
|
1167
|
+
const key = this.host.resolve(
|
|
1168
|
+
concat2(this.store.bytesPrefix(site.payload, ALL), tail),
|
|
1169
|
+
);
|
|
1170
|
+
if (key === null) continue;
|
|
1171
|
+
const nx = this.store.nextFirst(key, 1);
|
|
1172
|
+
if (nx.length === 0) continue;
|
|
1173
|
+
yield {
|
|
1174
|
+
premises: [fact],
|
|
1175
|
+
conclusion: {
|
|
1176
|
+
kind: "out",
|
|
1177
|
+
i: fact.i,
|
|
1178
|
+
j: queryLen,
|
|
1179
|
+
bytes: this.store.bytesPrefix(nx[0], ALL),
|
|
1180
|
+
cover: true,
|
|
1181
|
+
rec: true,
|
|
1182
|
+
node: nx[0],
|
|
1183
|
+
join: true,
|
|
1184
|
+
},
|
|
1185
|
+
cost: STEP,
|
|
1186
|
+
};
|
|
1187
|
+
}
|
|
1188
|
+
}
|
|
1189
|
+
|
|
1070
1190
|
/** out(i,j,bytes,…): index it for the binary rules, then offer splicing a
|
|
1071
1191
|
* learnt connector (the in-search bridge), splitting (at a sub-leaf form
|
|
1072
|
-
* boundary), bridging (cover(i) ∧ this → cover(j)),
|
|
1073
|
-
*
|
|
1192
|
+
* boundary), bridging (cover(i) ∧ this → cover(j)), fusing with an adjacent
|
|
1193
|
+
* finalised out, and — for a produced fact — JOINING the entity it contains
|
|
1194
|
+
* with the query's tail ({@link join}). */
|
|
1074
1195
|
private *outRules(
|
|
1075
1196
|
it: OutItem,
|
|
1076
1197
|
ctx: {
|
|
@@ -1087,6 +1208,8 @@ export class GraphSearch {
|
|
|
1087
1208
|
findBranchU: (k: number[]) => number | undefined;
|
|
1088
1209
|
linksByLeft?: ReadonlyMap<number, Array<[number, Uint8Array]>>;
|
|
1089
1210
|
linksByRight?: ReadonlyMap<number, Array<[number, Uint8Array]>>;
|
|
1211
|
+
queryBytes: Uint8Array;
|
|
1212
|
+
queryLen: number;
|
|
1090
1213
|
},
|
|
1091
1214
|
): Iterable<Rule<GItem>> {
|
|
1092
1215
|
const { splits, coversDone, outsByStart, outsByEnd, coverableByStart } =
|
|
@@ -1179,6 +1302,18 @@ export class GraphSearch {
|
|
|
1179
1302
|
|
|
1180
1303
|
for (const r of outsByStart.get(it.j) ?? []) yield* this.fuse(it, r, ctx);
|
|
1181
1304
|
for (const l of outsByEnd.get(it.i) ?? []) yield* this.fuse(l, it, ctx);
|
|
1305
|
+
|
|
1306
|
+
// ── JOIN (the A*LD extension) ───────────────────────────────────────
|
|
1307
|
+
// A produced fact may CONTAIN the subject the query never named; the query's
|
|
1308
|
+
// remaining tail then names the relation to follow FROM that subject. The
|
|
1309
|
+
// pair (contained entity, tail) is itself a learned key, and its
|
|
1310
|
+
// continuation is the derived answer — a genuine two-fact join, not the
|
|
1311
|
+
// juxtaposition the cover produces when the intermediate key IS named.
|
|
1312
|
+
// Fired per finalized out with a node, so it is the search's own rule, on
|
|
1313
|
+
// the ladder, memoised by {@link key}, and bounded by the fact's own length.
|
|
1314
|
+
if (it.node !== undefined) {
|
|
1315
|
+
yield* this.join(it, ctx.queryBytes, ctx.queryLen);
|
|
1316
|
+
}
|
|
1182
1317
|
}
|
|
1183
1318
|
|
|
1184
1319
|
/** Whether the query span [from, to) is wholly covered by RECOGNISED outs —
|
package/src/mind/recognition.ts
CHANGED
|
@@ -518,7 +518,21 @@ function recogniseImpl(ctx: MindContext, bytes: Uint8Array): Recognition {
|
|
|
518
518
|
end: number,
|
|
519
519
|
canonBudget: boolean,
|
|
520
520
|
): void => {
|
|
521
|
-
|
|
521
|
+
// Any span at least one river window wide is worth a probe. This used
|
|
522
|
+
// to stop at `chainReach(W)` — "the chain already covers anything that
|
|
523
|
+
// short" — and that premise is false for a NESTED form: the chain grows
|
|
524
|
+
// single-byte leaf ids and gates each step on `findBranch(ids)`, which
|
|
525
|
+
// is null for a form the write side chunked (measured: "Gustaf
|
|
526
|
+
// Molander" embedded in "The director of Eva is Gustaf Molander." gains
|
|
527
|
+
// no branch at any prefix, so `resolveSpan` is never reached), and its
|
|
528
|
+
// INTERIOR reach is one chunk plus W, which can be shorter than the
|
|
529
|
+
// form. The result was a dead zone: a form shorter than `chainReach`
|
|
530
|
+
// that neither starts on a fold cut nor ends on a node edge was
|
|
531
|
+
// unreachable by either tier — the exact site whose loss `tryChain`'s
|
|
532
|
+
// own note records as "the pivot dies with the site and multi-hop goes
|
|
533
|
+
// silent". The interior pass below spends the same budget on those
|
|
534
|
+
// pairs.
|
|
535
|
+
if (end - start < W) return;
|
|
522
536
|
if (flatProbe(start, end) === null) {
|
|
523
537
|
if (!canonBudget) return;
|
|
524
538
|
if (!canonAdmits(start, end)) return;
|
|
@@ -575,6 +589,24 @@ function recogniseImpl(ctx: MindContext, bytes: Uint8Array): Recognition {
|
|
|
575
589
|
if (i < prefixes.length && !spend(0, prefixes[i])) break;
|
|
576
590
|
if (i < suffixes.length && !spend(suffixes[i], bytes.length)) break;
|
|
577
591
|
}
|
|
592
|
+
// INTERIOR pairs, bounded by the same `chainReach(W)` the chain trusts —
|
|
593
|
+
// the dead zone the gate above used to leave: a form that neither starts
|
|
594
|
+
// on a fold cut nor ends on a node edge is exactly the one neither the
|
|
595
|
+
// chain (nested, `findBranch` misses) nor the two edge scans reach. The
|
|
596
|
+
// pair count is `|endpoints| · chainReach(W)`, i.e. LINEAR in the query —
|
|
597
|
+
// the W² span bound is what keeps this from being the quadratic scan the
|
|
598
|
+
// budget note above describes (that one had no span bound at all).
|
|
599
|
+
{
|
|
600
|
+
const reach = chainReach(W);
|
|
601
|
+
for (const end of ordered) {
|
|
602
|
+
for (const start of ordered) {
|
|
603
|
+
if (start >= end) continue;
|
|
604
|
+
const span = end - start;
|
|
605
|
+
if (span < W || span > reach) continue;
|
|
606
|
+
spend(start, end);
|
|
607
|
+
}
|
|
608
|
+
}
|
|
609
|
+
}
|
|
578
610
|
}
|
|
579
611
|
}
|
|
580
612
|
|
package/src/mind/trace.ts
CHANGED
|
@@ -75,6 +75,8 @@ export const MOVE_NOTE: Record<string, string> = {
|
|
|
75
75
|
"split": "cut a span at a sub-leaf form boundary so a form can be reached",
|
|
76
76
|
"fuse": "fuse adjacent fragments toward a deeper learned form",
|
|
77
77
|
"recompose": "recompose fused parts into a learned whole that leads on",
|
|
78
|
+
"join-fact":
|
|
79
|
+
"join a produced fact's own subject with the query's tail — derive through the fact, not alongside it",
|
|
78
80
|
"bridge": "advance the cover frontier across this span",
|
|
79
81
|
"pool-vote":
|
|
80
82
|
"pool independent regions' evidence for a shared anchor (sum, not shortest path)",
|
|
@@ -83,3 +83,33 @@ test("recognise(): a wide edge trim does not corrupt an unrelated short-form ans
|
|
|
83
83
|
const r = await m.respond("2+2 は何ですか");
|
|
84
84
|
assert.equal(dec(r.bytes), "4");
|
|
85
85
|
});
|
|
86
|
+
|
|
87
|
+
test("recognise(): an INTERIOR form at a non-cut offset is recovered, not only an edge one", async () => {
|
|
88
|
+
// The edge tier used to probe only prefixes of 0 and suffixes to bytes.length,
|
|
89
|
+
// and the flat-leaf chain cannot rebuild a write-side-chunked form (its
|
|
90
|
+
// `findBranch(ids)` pre-check misses at every prefix, so `resolveSpan` is
|
|
91
|
+
// never reached) while its interior reach is one chunk plus W. A form that
|
|
92
|
+
// neither starts on a fold cut nor ends on a node edge therefore fell in a
|
|
93
|
+
// dead zone — exactly the object inside a produced fact ("…is Gustaf
|
|
94
|
+
// Molander."), which is the site the pivot would need to chain on.
|
|
95
|
+
// MEASURED on the pre-change tree: no site for the entity. With the bounded
|
|
96
|
+
// interior pass (spans W..chainReach(W), linear in the query), it is found at
|
|
97
|
+
// its true span.
|
|
98
|
+
const m = new Mind({ seed: 7, store: new SQliteStore({ path: ":memory:" }) });
|
|
99
|
+
await m.ingest([
|
|
100
|
+
["x", "The director of Eva is Gustaf Molander."],
|
|
101
|
+
["Gustaf Molander", "The father of Gustaf Molander is Harald Molander."],
|
|
102
|
+
]);
|
|
103
|
+
const expected = resolve(m, enc("Gustaf Molander"));
|
|
104
|
+
assert.ok(expected !== null, "sanity: the entity must resolve standalone");
|
|
105
|
+
|
|
106
|
+
const rec = recognise(m, enc("The director of Eva is Gustaf Molander."));
|
|
107
|
+
const hit = rec.sites.find((s) => s.payload === expected);
|
|
108
|
+
assert.ok(
|
|
109
|
+
hit,
|
|
110
|
+
`expected an interior site for the entity, got: ` +
|
|
111
|
+
JSON.stringify(rec.sites.map((s) => [s.start, s.end, s.payload])),
|
|
112
|
+
);
|
|
113
|
+
assert.deepEqual([hit.start, hit.end], [23, 38]);
|
|
114
|
+
await m.store.close();
|
|
115
|
+
});
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
// 98-completion-chaining.test.mjs — the completion recursion NESTS, and the
|
|
2
|
+
// two properties that make that reachable are pinned here.
|
|
3
|
+
//
|
|
4
|
+
// `graph-search.ts`'s `recompleteNode` used to refuse any re-cover inside a
|
|
5
|
+
// re-cover (`recompleteOpen.size > 0`), so a chain through a PRODUCED composite
|
|
6
|
+
// stopped after a single recomposition: the depth the A*LD substrate licenses
|
|
7
|
+
// was unreachable. The recursion now nests, but only through ACCEPTED
|
|
8
|
+
// completions and only along the produced form's own parts — and with per-query
|
|
9
|
+
// work that stays the answer's (test/89 pins the cost).
|
|
10
|
+
//
|
|
11
|
+
// WHY THESE AND NOT test/15's §9–§12: those exercise chains whose parts all live
|
|
12
|
+
// INSIDE the query's own span ("a e", "x y", "p q r"); they pass even with
|
|
13
|
+
// `recompleteNode` disabled outright. §6 reaches a produced composite, but only
|
|
14
|
+
// one recomposition deep. The cases here fail on the pre-change tree:
|
|
15
|
+
// 1. the chain stops at the intermediate composite ("m n", not "z"), and
|
|
16
|
+
// 2. the nested derivation never reaches the rationale.
|
|
17
|
+
//
|
|
18
|
+
// MEASURED on the pre-change tree (this file's fixtures):
|
|
19
|
+
// respondText("seed") === "m n" (one recomposition)
|
|
20
|
+
// rationale moves = no `fuse`, no `recompose`
|
|
21
|
+
// After the change: "z", and both moves present.
|
|
22
|
+
|
|
23
|
+
import { test } from "node:test";
|
|
24
|
+
import assert from "node:assert/strict";
|
|
25
|
+
import { Mind } from "../dist/src/index.js";
|
|
26
|
+
|
|
27
|
+
/** seed → "p q" → (p→r, q→s) → "r s" → "m n" → (m→x, n→y) → "x y" → z
|
|
28
|
+
*
|
|
29
|
+
* The query names only `seed`. Every form after the first hop is PRODUCED by
|
|
30
|
+
* an edge, so reaching `z` requires TWO nested completions: "p q" recomposes to
|
|
31
|
+
* "m n", and "m n" recomposes again to "x y" → z. A single-level completion
|
|
32
|
+
* stops one composite short, at "m n". */
|
|
33
|
+
async function deepChain() {
|
|
34
|
+
const m = new Mind({ seed: 7 });
|
|
35
|
+
await m.ingest([
|
|
36
|
+
["seed", "p q"],
|
|
37
|
+
["p", "r"],
|
|
38
|
+
["q", "s"],
|
|
39
|
+
["r s", "m n"],
|
|
40
|
+
["m", "x"],
|
|
41
|
+
["n", "y"],
|
|
42
|
+
["x y", "z"],
|
|
43
|
+
]);
|
|
44
|
+
return m;
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
test("a produced composite completes through two nested recompositions", async () => {
|
|
48
|
+
const m = await deepChain();
|
|
49
|
+
assert.equal(
|
|
50
|
+
(await m.respondText("seed")).replace(/\0+/g, ""),
|
|
51
|
+
"z",
|
|
52
|
+
);
|
|
53
|
+
await m.store.close();
|
|
54
|
+
});
|
|
55
|
+
|
|
56
|
+
test("nested completions reach the rationale", async () => {
|
|
57
|
+
const m = await deepChain();
|
|
58
|
+
const steps = [];
|
|
59
|
+
await m.respondText("seed", (s) => steps.push(s));
|
|
60
|
+
const moves = new Set(steps.map((s) => s.mechanism.at(-1)));
|
|
61
|
+
// The top cover cannot fabricate these on its own: recomposition happens
|
|
62
|
+
// inside the nested solves, and they were invisible before the sink was
|
|
63
|
+
// threaded through.
|
|
64
|
+
assert.ok(
|
|
65
|
+
moves.has("recompose"),
|
|
66
|
+
`expected a recompose move in the rationale, got: ${[...moves].join(", ")}`,
|
|
67
|
+
);
|
|
68
|
+
assert.ok(
|
|
69
|
+
moves.has("fuse"),
|
|
70
|
+
`expected a fuse move in the rationale, got: ${[...moves].join(", ")}`,
|
|
71
|
+
);
|
|
72
|
+
await m.store.close();
|
|
73
|
+
});
|
|
74
|
+
|
|
75
|
+
test("a produced composite is completed by its own parts, not by a learned form inside it", async () => {
|
|
76
|
+
// The produced bytes are "hi there"; `hi` is a learned context that is NOT one
|
|
77
|
+
// of their parts. Re-recognising arbitrary forms inside the bytes is what
|
|
78
|
+
// let a hub-heavy utterance explode on the trained store (`respond("hi.")`:
|
|
79
|
+
// 37 bytes, seven hub openers, 2.5 GB, OOM) — the produced form is decomposed
|
|
80
|
+
// instead by the machinery that owns its shape (leaves/splits), with the
|
|
81
|
+
// recognised SITES filtered to the node's own kids.
|
|
82
|
+
//
|
|
83
|
+
// MEASURED on the pre-change tree: no `recompose` move here at all — the
|
|
84
|
+
// unrelated `hi` site was taken instead of decomposing the form.
|
|
85
|
+
const m = new Mind({ seed: 7 });
|
|
86
|
+
await m.ingest([
|
|
87
|
+
["seed", "hi there"],
|
|
88
|
+
["hi", "KLX"],
|
|
89
|
+
]);
|
|
90
|
+
const steps = [];
|
|
91
|
+
const answer = (await m.respondText("seed", (s) => steps.push(s)))
|
|
92
|
+
.replace(/\0+/g, "")
|
|
93
|
+
.trim();
|
|
94
|
+
assert.equal(answer, "hi there");
|
|
95
|
+
const moves = new Set(steps.map((s) => s.mechanism.at(-1)));
|
|
96
|
+
assert.ok(
|
|
97
|
+
moves.has("recompose"),
|
|
98
|
+
`expected the completion to decompose the form by its parts, got: ${
|
|
99
|
+
[...moves].join(", ")
|
|
100
|
+
}`,
|
|
101
|
+
);
|
|
102
|
+
await m.store.close();
|
|
103
|
+
});
|
|
104
|
+
|
|
105
|
+
test("the chain runs as deep as the graph licenses (four nested completions)", async () => {
|
|
106
|
+
const m = new Mind({ seed: 7 });
|
|
107
|
+
await m.ingest([
|
|
108
|
+
["seed", "p1 q1"],
|
|
109
|
+
["p1", "a1"],
|
|
110
|
+
["q1", "b1"],
|
|
111
|
+
["a1 b1", "p2 q2"],
|
|
112
|
+
["p2", "a2"],
|
|
113
|
+
["q2", "b2"],
|
|
114
|
+
["a2 b2", "p3 q3"],
|
|
115
|
+
["p3", "a3"],
|
|
116
|
+
["q3", "b3"],
|
|
117
|
+
["a3 b3", "FIM"],
|
|
118
|
+
]);
|
|
119
|
+
assert.equal((await m.respondText("seed")).replace(/\0+/g, ""), "FIM");
|
|
120
|
+
await m.store.close();
|
|
121
|
+
});
|
|
122
|
+
|
|
123
|
+
test("a completion cycle terminates deterministically (the stack is the guard)", async () => {
|
|
124
|
+
// The recomposition of "x y" leads back to "x y" itself. Membership in
|
|
125
|
+
// `recompleteOpen` refuses the re-entry, so the recursion terminates on the
|
|
126
|
+
// form it already holds instead of re-covering it forever — and it answers
|
|
127
|
+
// the same bytes twice.
|
|
128
|
+
const m = new Mind({ seed: 7 });
|
|
129
|
+
await m.ingest([
|
|
130
|
+
["seed", "x y"],
|
|
131
|
+
["x", "A"],
|
|
132
|
+
["y", "B"],
|
|
133
|
+
["A B", "x y"],
|
|
134
|
+
]);
|
|
135
|
+
const first = (await m.respondText("seed")).replace(/\0+/g, "").trim();
|
|
136
|
+
const second = (await m.respondText("seed")).replace(/\0+/g, "").trim();
|
|
137
|
+
assert.equal(first, second);
|
|
138
|
+
assert.equal(first, "x y");
|
|
139
|
+
await m.store.close();
|
|
140
|
+
});
|
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
// 99-fact-join.test.mjs — the JOIN: derive the answer THROUGH a produced fact
|
|
2
|
+
// whose subject the query never names.
|
|
3
|
+
//
|
|
4
|
+
// THE MOVE THIS PINS. A query like "Eiffel Tower country capital" names hop 1's
|
|
5
|
+
// key ("Eiffel Tower country") and hop 2's relation (" capital"), but never the
|
|
6
|
+
// intermediate subject ("France"). The cover alone can only juxtapose the two
|
|
7
|
+
// facts; the pivot needs the subject to surface as an unconsumed site inside the
|
|
8
|
+
// produced fact. The `join` rule in graph-search.ts closes the gap: it takes the
|
|
9
|
+
// entity the produced fact CONTAINS, combines it with the query's remaining tail,
|
|
10
|
+
// and asks the store for that key's continuation — a direct STEP on the ladder,
|
|
11
|
+
// reported as `join-fact` in the rationale.
|
|
12
|
+
//
|
|
13
|
+
// WHY THE FIXTURE CROSSES N=4096. Below the atomIsHub flip the interior subject
|
|
14
|
+
// surfaces anyway (small-store recognition) and the PIVOT alone reaches the
|
|
15
|
+
// chain, so a small fixture passes under the pre-change tree too — measured on a
|
|
16
|
+
// 3-deposit store, which answered "The father of Gustaf Molander is Harald
|
|
17
|
+
// Molander." with no join. The join is what carries the chain at corpus scale,
|
|
18
|
+
// so the fixture must sit past the flip, exactly as test/78 does. The filler is
|
|
19
|
+
// lexically varied for the same reason test/78's is: a templated corpus folds to
|
|
20
|
+
// shared chunks and leaves the query uncontested.
|
|
21
|
+
|
|
22
|
+
import { test } from "node:test";
|
|
23
|
+
import assert from "node:assert/strict";
|
|
24
|
+
import { Mind } from "../dist/src/index.js";
|
|
25
|
+
import { SQliteStore } from "../dist/src/store-sqlite.js";
|
|
26
|
+
|
|
27
|
+
const CHAIN = [
|
|
28
|
+
["Eiffel Tower country", "The country of Eiffel Tower is France."],
|
|
29
|
+
["France capital", "The capital of France is Paris."],
|
|
30
|
+
// The pivot fact: the intermediate subject as a context of its own.
|
|
31
|
+
["France", "The capital of France is Paris."],
|
|
32
|
+
];
|
|
33
|
+
|
|
34
|
+
const WORDS =
|
|
35
|
+
("alpha bravo charlie delta echo foxtrot golf hotel india juliet " +
|
|
36
|
+
"kilo lima mike november oscar papa quebec romeo sierra tango uniform " +
|
|
37
|
+
"victor whiskey xray yankee zulu amber bronze copper dahlia ember fjord " +
|
|
38
|
+
"gossamer harbour indigo jasmine kestrel lantern marigold nectar opal " +
|
|
39
|
+
"pewter quartz ripple saffron thistle umber violet willow xenon yarrow")
|
|
40
|
+
.split(" ");
|
|
41
|
+
const filler = (i) => {
|
|
42
|
+
const w = (n) => WORDS[(i * 7 + n * 13) % WORDS.length];
|
|
43
|
+
return [
|
|
44
|
+
`${w(1)} ${w(2)} ${w(3)} ${i}`,
|
|
45
|
+
`${w(4)} ${w(5)} ${w(6)} ${w(7)} ${i}`,
|
|
46
|
+
];
|
|
47
|
+
};
|
|
48
|
+
|
|
49
|
+
/** One store, ingested past the atomIsHub flip. */
|
|
50
|
+
async function pastTheFlip() {
|
|
51
|
+
const store = new SQliteStore({ path: ":memory:", D: 1024 });
|
|
52
|
+
const mind = new Mind({ seed: 7, store });
|
|
53
|
+
await mind.ingest(CHAIN);
|
|
54
|
+
await mind.ingest(Array.from({ length: 4300 }, (_, i) => filler(i)));
|
|
55
|
+
return { store, mind };
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
test("a join derives through a produced fact whose subject the query never names", async () => {
|
|
59
|
+
const { store, mind } = await pastTheFlip();
|
|
60
|
+
const query = "Eiffel Tower country capital";
|
|
61
|
+
assert.ok(
|
|
62
|
+
!query.includes("France") && !query.includes("capital of France"),
|
|
63
|
+
"sanity: the intermediate subject must NOT be named in the query",
|
|
64
|
+
);
|
|
65
|
+
|
|
66
|
+
const moves = [];
|
|
67
|
+
const out = await mind.respondText(
|
|
68
|
+
query,
|
|
69
|
+
(s) => moves.push(s.mechanism[s.mechanism.length - 1]),
|
|
70
|
+
);
|
|
71
|
+
|
|
72
|
+
assert.equal(out.trim(), "The capital of France is Paris.");
|
|
73
|
+
assert.ok(
|
|
74
|
+
moves.includes("join-fact"),
|
|
75
|
+
`expected the join to be the move that reached the answer, got: ${
|
|
76
|
+
[...new Set(moves)].join(", ")
|
|
77
|
+
}`,
|
|
78
|
+
);
|
|
79
|
+
await store.close();
|
|
80
|
+
});
|
|
81
|
+
|
|
82
|
+
test("naming the intermediate does not need the join — the cover reads it directly", async () => {
|
|
83
|
+
// The contrast that keeps the move honest: when the subject IS written, the
|
|
84
|
+
// cover already reaches the chain, so no join is claimed.
|
|
85
|
+
const { store, mind } = await pastTheFlip();
|
|
86
|
+
const moves = [];
|
|
87
|
+
const out = await mind.respondText(
|
|
88
|
+
"Eiffel Tower country France capital",
|
|
89
|
+
(s) => moves.push(s.mechanism[s.mechanism.length - 1]),
|
|
90
|
+
);
|
|
91
|
+
assert.ok(out.includes("The capital of France is Paris."));
|
|
92
|
+
assert.ok(
|
|
93
|
+
!moves.includes("join-fact"),
|
|
94
|
+
"a named intermediate must not be reported as a join",
|
|
95
|
+
);
|
|
96
|
+
await store.close();
|
|
97
|
+
});
|