@hviana/sema 0.8.0 → 0.8.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +22 -1
- package/DATASETS.md +1 -1
- package/dist/example/train_base/config.js +2 -2
- package/dist/example/train_base/corpora/massive.js +1 -1
- package/dist/example/train_base/readers.js +1 -1
- package/dist/src/config.d.ts +17 -0
- package/dist/src/config.js +18 -0
- package/dist/src/geometry.d.ts +10 -10
- package/dist/src/geometry.js +25 -24
- package/dist/src/meter.d.ts +29 -12
- package/dist/src/meter.js +58 -14
- package/dist/src/mind/attention.js +12 -12
- package/dist/src/mind/bridge.d.ts +8 -8
- package/dist/src/mind/bridge.js +33 -32
- package/dist/src/mind/corpus.d.ts +40 -0
- package/dist/src/mind/corpus.js +149 -0
- package/dist/src/mind/graph-search.d.ts +7 -8
- package/dist/src/mind/graph-search.js +244 -32
- package/dist/src/mind/index.d.ts +3 -1
- package/dist/src/mind/index.js +1 -0
- package/dist/src/mind/junction.d.ts +1 -1
- package/dist/src/mind/junction.js +8 -8
- package/dist/src/mind/learning.js +36 -35
- package/dist/src/mind/match.d.ts +8 -3
- package/dist/src/mind/match.js +156 -71
- package/dist/src/mind/mechanisms/cast.js +18 -2
- package/dist/src/mind/mechanisms/cover.js +19 -12
- package/dist/src/mind/mechanisms/prefix-completion.js +24 -24
- package/dist/src/mind/mechanisms/recall.js +38 -40
- package/dist/src/mind/mechanisms/reference.js +16 -16
- package/dist/src/mind/mind.d.ts +61 -7
- package/dist/src/mind/mind.js +72 -2
- package/dist/src/mind/pipeline-mechanism.d.ts +10 -8
- package/dist/src/mind/pipeline-mechanism.js +25 -21
- package/dist/src/mind/pipeline.d.ts +9 -9
- package/dist/src/mind/pipeline.js +49 -29
- package/dist/src/mind/primitives.d.ts +5 -5
- package/dist/src/mind/primitives.js +5 -5
- package/dist/src/mind/reasoning.d.ts +5 -1
- package/dist/src/mind/reasoning.js +54 -1
- package/dist/src/mind/recognition.d.ts +14 -13
- package/dist/src/mind/recognition.js +23 -23
- package/dist/src/mind/resonance.js +21 -21
- package/dist/src/mind/traverse.d.ts +54 -52
- package/dist/src/mind/traverse.js +83 -73
- package/dist/src/mind/types.d.ts +26 -4
- package/dist/src/store.d.ts +12 -12
- package/dist/src/store.js +12 -12
- package/docs/INDEX.md +2 -2
- package/docs/architecture/exact-vs-approximate.md +2 -1
- package/docs/architecture/fold-contract.md +1 -1
- package/docs/failures/tempting-but-wrong.md +33 -5
- package/docs/harness/gates.md +7 -7
- package/example/train_base/config.ts +2 -2
- package/example/train_base/corpora/massive.ts +1 -1
- package/example/train_base/readers.ts +1 -1
- package/jsr.json +1 -1
- package/package.json +1 -1
- package/src/config.ts +35 -0
- package/src/geometry.ts +25 -24
- package/src/meter.ts +61 -14
- package/src/mind/attention.ts +12 -12
- package/src/mind/bridge.ts +33 -32
- package/src/mind/corpus.ts +202 -0
- package/src/mind/graph-search.ts +261 -31
- package/src/mind/index.ts +8 -1
- package/src/mind/junction.ts +8 -8
- package/src/mind/learning.ts +36 -35
- package/src/mind/match.ts +163 -73
- package/src/mind/mechanisms/cast.ts +17 -1
- package/src/mind/mechanisms/cover.ts +18 -12
- package/src/mind/mechanisms/prefix-completion.ts +24 -24
- package/src/mind/mechanisms/recall.ts +38 -40
- package/src/mind/mechanisms/reference.ts +16 -16
- package/src/mind/mind.ts +129 -7
- package/src/mind/pipeline-mechanism.ts +25 -21
- package/src/mind/pipeline.ts +63 -38
- package/src/mind/primitives.ts +5 -5
- package/src/mind/reasoning.ts +55 -0
- package/src/mind/recognition.ts +23 -23
- package/src/mind/resonance.ts +21 -21
- package/src/mind/traverse.ts +83 -73
- package/src/mind/types.ts +30 -4
- package/src/store.ts +20 -20
- package/test/08-storage.test.mjs +1 -1
- package/test/100-complete-grounding-trace.test.mjs +109 -0
- package/test/101-alignment-gap-bound.test.mjs +106 -0
- package/test/102-production-composes-at-scale.test.mjs +110 -0
- package/test/103-alignment-gap-budget.test.mjs +89 -0
- package/test/104-composition-is-reported.test.mjs +90 -0
- package/test/105-derive-through-reports-its-refusal.test.mjs +113 -0
- package/test/106-the-join-fires.test.mjs +94 -0
- package/test/107-the-join-is-counted.test.mjs +81 -0
- package/test/108-the-join-chains.test.mjs +78 -0
- package/test/109-the-pivot-is-counted.test.mjs +60 -0
- package/test/110-the-reasoner-stops-when-the-question-is-answered.test.mjs +91 -0
- package/test/111-the-cover-assembly-is-counted.test.mjs +74 -0
- package/test/112-the-exploration-does-not-grow-with-the-hub.test.mjs +89 -0
- package/test/113-the-rationale-payload-is-bounded.test.mjs +84 -0
- package/test/114-alignment-budget-is-per-sweep.test.mjs +93 -0
- package/test/116-the-extension-is-gated-by-the-pipelines-own-remainder.test.mjs +100 -0
- package/test/117-corpus-search.test.mjs +171 -0
- package/test/14-scaling.test.mjs +10 -7
- package/test/35-prefix-edge.test.mjs +1 -1
- package/test/40-choosenext-scale-guard.test.mjs +16 -17
- package/test/56-bridge-identity-admission.test.mjs +6 -6
- package/test/70-prefix-completion.test.mjs +4 -3
- package/test/72-prefix-candidate-supply.test.mjs +3 -3
- package/test/73-scaffolding-only-bridge-abstains.test.mjs +6 -6
- package/test/75-multiturn-context-optimisation.test.mjs +5 -5
- package/test/76-reference-binding.test.mjs +6 -1
- package/test/84-composed-answer-honesty.test.mjs +5 -6
- package/test/88-dependency-footprint.test.mjs +1 -1
- package/test/89-completion-recursion.test.mjs +47 -19
- package/test/90-connector-read-cap.test.mjs +10 -8
- package/test/93-regime-prediction.test.mjs +10 -10
- package/test/94-cross-region-budget.test.mjs +2 -2
- package/test/95-wide-resonance-removed.test.mjs +8 -7
- package/test/96-bytes-walk-termination.test.mjs +3 -3
package/src/mind/resonance.ts
CHANGED
|
@@ -380,15 +380,15 @@ export async function pivotInto(
|
|
|
380
380
|
// Byte containment, longest wins — the answer literally contains the
|
|
381
381
|
// pivot's bytes, and the biggest well-evidenced span is the real pivot.
|
|
382
382
|
//
|
|
383
|
-
// REAL SATURATION, not a hard cap: the score IS the candidate's byte
|
|
384
|
-
//
|
|
385
|
-
//
|
|
386
|
-
//
|
|
387
|
-
// the cheap ordering key, and the first-inserted tie-break is made
|
|
388
|
-
// (`a.index - b.index`) so equal lengths keep `scored`'s insertion
|
|
389
|
-
// exactly the tie argmaxBy(strict) used to keep.
|
|
390
|
-
// winning candidate are read; every shorter candidate the probes proposed
|
|
391
|
-
// skipped without reconstruction, where the old argmax read them all.
|
|
383
|
+
// REAL SATURATION, not a hard cap: the score IS the candidate's byte length,
|
|
384
|
+
// so the scan is DECIDED the moment the first candidate that passes every
|
|
385
|
+
// filter is found in DESCENDING length order — a shorter candidate can never
|
|
386
|
+
// outscore it. `contentLen` (the prefix-capped length read, bounded-reads.md)
|
|
387
|
+
// is the cheap ordering key, and the first-inserted tie-break is made
|
|
388
|
+
// explicit (`a.index - b.index`) so equal lengths keep `scored`'s insertion
|
|
389
|
+
// order — exactly the tie argmaxBy(strict) used to keep. The bytes of at most
|
|
390
|
+
// ONE winning candidate are read; every shorter candidate the probes proposed
|
|
391
|
+
// is skipped without reconstruction, where the old argmax read them all.
|
|
392
392
|
const ranked = [...scored.keys()]
|
|
393
393
|
.map((id, index) => ({
|
|
394
394
|
id,
|
|
@@ -399,11 +399,11 @@ export async function pivotInto(
|
|
|
399
399
|
let pivotId: number | null = null;
|
|
400
400
|
for (const c of ranked) {
|
|
401
401
|
const id = c.id;
|
|
402
|
-
// A ZERO-LENGTH candidate is not a pivot.
|
|
402
|
+
// A ZERO-LENGTH candidate is not a pivot. `argmaxBy(…, 0, strict)` used to
|
|
403
403
|
// carry this floor in its threshold argument, and dropping it here would
|
|
404
404
|
// admit an empty node: `indexOf(answer, <empty>)` returns 0, so every
|
|
405
|
-
// filter below passes and the chain would hop through nothing
|
|
406
|
-
// empty bytes are truthy).
|
|
405
|
+
// filter below passes and the chain would hop through nothing
|
|
406
|
+
// (INVARIANTS.md — empty bytes are truthy).
|
|
407
407
|
if (c.len === 0) continue;
|
|
408
408
|
// A PIVOT MUST BE A THING THE CORPUS DEPOSITED, NOT A PIECE OF ONE.
|
|
409
409
|
// "Longest wins" ranks candidates but never asks whether the winner is
|
|
@@ -439,15 +439,15 @@ export async function pivotInto(
|
|
|
439
439
|
// a span that was never a fact on its own is not one to step through.
|
|
440
440
|
// No constant enters — it is a structural predicate, not a threshold.
|
|
441
441
|
if (ctx.store.hasParents(id) || ctx.store.hasContainers(id)) continue;
|
|
442
|
-
// A candidate whose bytes are LONGER than the answer cannot be a
|
|
443
|
-
//
|
|
444
|
-
//
|
|
445
|
-
//
|
|
446
|
-
//
|
|
447
|
-
//
|
|
448
|
-
//
|
|
449
|
-
//
|
|
450
|
-
//
|
|
442
|
+
// A candidate whose bytes are LONGER than the answer cannot be a substring
|
|
443
|
+
// of it — `indexOf` would return −1 regardless. Prune by length BEFORE
|
|
444
|
+
// reconstructing the bytes: `read` is an UNCAPPED read (bounded-reads.md),
|
|
445
|
+
// and a resonated context far longer than the answer is exactly the
|
|
446
|
+
// candidate that makes it cost a whole deposit's worth of reconstruction
|
|
447
|
+
// for a containment test that must fail. `contentLen` with the
|
|
448
|
+
// `answer.length + 1` cap is the prefix-capped length read the same
|
|
449
|
+
// contract prescribes; the prune is byte-identical to the old `indexOf`
|
|
450
|
+
// miss (it returns −1 for a needle longer than the haystack).
|
|
451
451
|
if (c.len > answer.length) continue;
|
|
452
452
|
const bytes = read(ctx, id);
|
|
453
453
|
if (indexOf(answer, bytes, 0) < 0) continue;
|
package/src/mind/traverse.ts
CHANGED
|
@@ -32,11 +32,11 @@ interface StructCache {
|
|
|
32
32
|
hasParents: Map<number, boolean>;
|
|
33
33
|
}
|
|
34
34
|
//
|
|
35
|
-
// Budgeted on the same terms as the reach memo below (
|
|
36
|
-
//
|
|
37
|
-
//
|
|
38
|
-
//
|
|
39
|
-
//
|
|
35
|
+
// Budgeted on the same terms as the reach memo below (caches.md): these three
|
|
36
|
+
// maps are cleared on every write, but a long read-only session over a large
|
|
37
|
+
// store converges on one entry per node per map with nothing to bound it. Past
|
|
38
|
+
// the cap all three are dropped together and re-derived, costing cold
|
|
39
|
+
// structural probes and never a wrong answer.
|
|
40
40
|
const STRUCT_MEMO_MAX = 100_000;
|
|
41
41
|
const structCaches = new WeakMap<object, StructCache>();
|
|
42
42
|
|
|
@@ -55,21 +55,22 @@ const structCaches = new WeakMap<object, StructCache>();
|
|
|
55
55
|
// battery repeatedly reaches the same corpus scaffolding even when its
|
|
56
56
|
// surface questions differ.
|
|
57
57
|
//
|
|
58
|
-
// Budgeted, not unbounded (
|
|
59
|
-
//
|
|
58
|
+
// Budgeted, not unbounded (caches.md): past the cap the whole map is dropped
|
|
59
|
+
// and
|
|
60
|
+
// re-derived, costing a cold climb and never a wrong answer.
|
|
60
61
|
const REACH_MEMO_MAX = 100_000;
|
|
61
62
|
const reachCaches = new WeakMap<object, Map<number, AncestorReach>>();
|
|
62
63
|
|
|
63
64
|
/** The reach memo this ask should use — see the note above.
|
|
64
65
|
*
|
|
65
|
-
* A TRACED response always gets a fresh, empty one.
|
|
66
|
-
*
|
|
67
|
-
*
|
|
68
|
-
*
|
|
69
|
-
*
|
|
70
|
-
*
|
|
71
|
-
*
|
|
72
|
-
*
|
|
66
|
+
* A TRACED response always gets a fresh, empty one. `AncestorReach`'s
|
|
67
|
+
* `visited`/`maxDepth`/`saturation` fields are populated only when a trace is
|
|
68
|
+
* attached, so an entry deposited by an untraced earlier turn would silently
|
|
69
|
+
* black out the reach detail of a later traced one; and the trace's reach
|
|
70
|
+
* payload is serialised by ITERATING this map, which must therefore hold what
|
|
71
|
+
* THIS climb consulted, not the whole conversation's history. Consistent with
|
|
72
|
+
* memoization.md: a traced response is a different machine — never benchmark
|
|
73
|
+
* with a trace attached. */
|
|
73
74
|
export function sharedReachMemo(
|
|
74
75
|
ctx: MindContext,
|
|
75
76
|
): Map<number, AncestorReach> {
|
|
@@ -481,13 +482,14 @@ export function bearsEdge(ctx: MindContext, id: number): boolean {
|
|
|
481
482
|
}
|
|
482
483
|
|
|
483
484
|
/** Whether a node LEADS SOMEWHERE — it bears a continuation edge or a halo.
|
|
484
|
-
*
|
|
485
|
-
*
|
|
486
|
-
*
|
|
487
|
-
*
|
|
488
|
-
*
|
|
489
|
-
*
|
|
490
|
-
*
|
|
485
|
+
* The admission predicate recognition filters sites with (cover.md): a form
|
|
486
|
+
* that
|
|
487
|
+
* leads nowhere contributes nothing to any derivation. Runs once per candidate
|
|
488
|
+
* span on the recognition hot path — `hasNext` is cached per response (the same
|
|
489
|
+
* flat-branch ids are probed across prefix variants by canonicalChunkId).
|
|
490
|
+
* `hasHalo` is not cached: it's a single indexed point probe per candidate, and
|
|
491
|
+
* the candidates that reach this check have already been filtered by hasNext
|
|
492
|
+
* above in edgeAncestors. */
|
|
491
493
|
export function leadsSomewhere(ctx: MindContext, id: number): boolean {
|
|
492
494
|
const memo = getStructCache(ctx);
|
|
493
495
|
if (cachedHasNext(ctx, id, memo)) return true;
|
|
@@ -539,10 +541,10 @@ function boundFor(contextCount: number): number {
|
|
|
539
541
|
}
|
|
540
542
|
|
|
541
543
|
/** Cap a candidate list at the hub bound √N (insertion order) — the ONE
|
|
542
|
-
*
|
|
543
|
-
*
|
|
544
|
-
*
|
|
545
|
-
*
|
|
544
|
+
* fan-out convention every walk and disambiguation uses (see bounded-reads.md).
|
|
545
|
+
* A node connected to more than √N others is a hub whose individual connections
|
|
546
|
+
* carry ~no discriminative information; materialising or scoring them all would
|
|
547
|
+
* make single decisions scale with the corpus. */
|
|
546
548
|
export function hubCap<T>(
|
|
547
549
|
ctx: MindContext,
|
|
548
550
|
ids: readonly T[],
|
|
@@ -581,16 +583,16 @@ export function contains(
|
|
|
581
583
|
* the EXACT half's veto on calling them synonyms.
|
|
582
584
|
*
|
|
583
585
|
* Halos measure company, and the strongest company any two forms can keep is
|
|
584
|
-
*
|
|
585
|
-
*
|
|
586
|
-
*
|
|
587
|
-
*
|
|
588
|
-
*
|
|
589
|
-
*
|
|
590
|
-
*
|
|
591
|
-
*
|
|
592
|
-
*
|
|
593
|
-
*
|
|
586
|
+
* standing next to each other: a question and its answer co-occur in every
|
|
587
|
+
* episode that taught the pair, so their halos SHOULD be similar, and on a
|
|
588
|
+
* conversational store they are (measured on the CONV fixture: consecutive
|
|
589
|
+
* turns at 0.809 against a 0.516 concept threshold). A gate reading halo cosine
|
|
590
|
+
* alone therefore reads adjacency as synonymy and revoices an answer in the
|
|
591
|
+
* words of the question it answers — "it hangs in madrid" spliced back into
|
|
592
|
+
* "where is it kept now". The distributional layer cannot tell the two
|
|
593
|
+
* relations apart, because to it they are the same observation; the exact half
|
|
594
|
+
* can, for free, because it stored the edge. halo-sketch.md's division of
|
|
595
|
+
* labour exactly: approximate proposes, exact decides.
|
|
594
596
|
*
|
|
595
597
|
* Read LIMITed in both directions at the hub bound — a common continuation's
|
|
596
598
|
* fan-in is corpus-sized, and no single decision may scale with it. */
|
|
@@ -745,26 +747,33 @@ export function chooseNext(
|
|
|
745
747
|
// NO consensusFloor gate here (tried and reverted — see
|
|
746
748
|
// test/40-choosenext-scale-guard.test.mjs): that floor is calibrated for
|
|
747
749
|
// POOLED, IDF-weighted CLIMB VOTES (recallByResonance, commitVotes), where
|
|
748
|
-
// each corroborating region contributes at most ln N and the floor grows
|
|
749
|
-
//
|
|
750
|
-
//
|
|
751
|
-
//
|
|
752
|
-
//
|
|
753
|
-
// N-
|
|
754
|
-
//
|
|
755
|
-
//
|
|
756
|
-
//
|
|
757
|
-
//
|
|
758
|
-
//
|
|
759
|
-
//
|
|
760
|
-
// chooseNext pseudocode, which has no such floor.
|
|
750
|
+
// each corroborating region contributes at most ln N and the floor grows with
|
|
751
|
+
// N exactly as that per-region ceiling does (thresholds.md). `bestSupport`
|
|
752
|
+
// here is a different kind of quantity — a raw prevCount of how many training
|
|
753
|
+
// contexts predicted ONE destination, bounded by how often that specific fact
|
|
754
|
+
// was retold, never by corpus size N. Gating an N-invariant count against an
|
|
755
|
+
// N-growing threshold guarantees failure once N is large enough, discarding
|
|
756
|
+
// genuinely, structurally dominant edges (observed: a fact corroborated
|
|
757
|
+
// 2-to-1-1-1 refused at N≈325K, falling back to a noisy concept-hop). The
|
|
758
|
+
// loop above already IS the "genuinely competing" test: a tie leaves
|
|
759
|
+
// first-inserted as the pick (test/30's own pinned behaviour); a strict
|
|
760
|
+
// winner is real evidence regardless of corpus scale. Matches `chooseNext`'s
|
|
761
|
+
// own pseudocode, which has no such floor.
|
|
761
762
|
|
|
762
763
|
// Trace is built lazily — the filter + map below only execute when a
|
|
763
764
|
// trace listener is attached, so the common (no-trace) path pays only
|
|
764
765
|
// for the prevCount calls in the loop above, never for extra rItemShort
|
|
765
766
|
// byte-reads.
|
|
766
767
|
if (ctx.trace) {
|
|
767
|
-
|
|
768
|
+
// A BOUNDED SAMPLE, AND THE COUNT. The step used to carry EVERY candidate
|
|
769
|
+
// it weighed — measured on the trained store, 1559 out-items in one step
|
|
770
|
+
// (hubBound's own size) and 1082 in another (the hub's degree). The
|
|
771
|
+
// rationale's job is to explain the CHOICE, and the count is what says how
|
|
772
|
+
// wide the field was; the declared candidate budget (`recallQueryK`) is what
|
|
773
|
+
// bounds the sample, so no number is invented here.
|
|
774
|
+
const others = capped
|
|
775
|
+
.filter((c) => c !== best)
|
|
776
|
+
.slice(0, ctx.cfg.rationaleSampleK);
|
|
768
777
|
ctx.trace.step(
|
|
769
778
|
"disambiguate",
|
|
770
779
|
[rItemShort(ctx, best, "halo-evidence", bestSupport)],
|
|
@@ -838,12 +847,13 @@ function rItemShort(
|
|
|
838
847
|
* W-window it spells is contained by more places than the hub bound allows,
|
|
839
848
|
* i.e. the whole query is corpus-global scaffolding.
|
|
840
849
|
*
|
|
841
|
-
*
|
|
842
|
-
*
|
|
843
|
-
*
|
|
844
|
-
*
|
|
845
|
-
*
|
|
846
|
-
*
|
|
850
|
+
* WHAT IT IS FOR. Several mechanisms ground a query through the literal spans
|
|
851
|
+
* it
|
|
852
|
+
* did NOT explain, and those spans are the whole of their evidence. When every
|
|
853
|
+
* one of them is a hub, the query says nothing the corpus can be held to, and
|
|
854
|
+
* grounding it means picking one of thousands of continuations it gives no
|
|
855
|
+
* evidence for — a fabrication whatever the answer happens to be. Answering
|
|
856
|
+
* with silence there is the honest degradation contract (INVARIANTS.md).
|
|
847
857
|
*
|
|
848
858
|
* MEASURED SEPARATION (trained store, hubBound 571) — this is categorical,
|
|
849
859
|
* not marginal, and it is why the predicate lives here rather than being
|
|
@@ -860,11 +870,11 @@ function rItemShort(
|
|
|
860
870
|
* evidence and sit on the SAME side as the correct ones, so this predicate
|
|
861
871
|
* is not what makes them silent and cannot be credited for them.
|
|
862
872
|
*
|
|
863
|
-
* NO NEW THRESHOLD (
|
|
864
|
-
*
|
|
865
|
-
*
|
|
866
|
-
*
|
|
867
|
-
*
|
|
873
|
+
* NO NEW THRESHOLD (thresholds.md): `hubBound` is the √N reading of "hub" used
|
|
874
|
+
* everywhere, and the containment read is clamped to it exactly as every other
|
|
875
|
+
* fan-out read is (bounded-reads.md). A query with no stored window at all is
|
|
876
|
+
* NOT scaffolding-only — it has no evidence either way, and its callers already
|
|
877
|
+
* refuse it on their own terms. */
|
|
868
878
|
export function allWindowsAreScaffolding(
|
|
869
879
|
ctx: MindContext,
|
|
870
880
|
query: Uint8Array,
|
|
@@ -916,19 +926,19 @@ export function allWindowsAreScaffolding(
|
|
|
916
926
|
* by climbing containment then parents. Nothing is added to the write side;
|
|
917
927
|
* this reads an index training already built.
|
|
918
928
|
*
|
|
919
|
-
* BOUNDED (
|
|
920
|
-
*
|
|
921
|
-
*
|
|
922
|
-
*
|
|
923
|
-
*
|
|
924
|
-
*
|
|
925
|
-
*
|
|
926
|
-
*
|
|
927
|
-
*
|
|
929
|
+
* BOUNDED (bounded-reads.md), AND WITH NO NEW THRESHOLD. The window whose
|
|
930
|
+
* containment is SMALLEST carries the most evidence, and one saturated at
|
|
931
|
+
* `hubBound` carries none — that is the same √N reading of "hub" the rest of
|
|
932
|
+
* the mind uses, not a tuned knob. The upward walk spends a budget of
|
|
933
|
+
* `hubBound` nodes and fans out by W, so a hub query enumerates nothing and the
|
|
934
|
+
* caller stays silent rather than guessing (INVARIANTS.md). Measured on the
|
|
935
|
+
* trained store: the photosynthesis form at a one-byte truncation picks a
|
|
936
|
+
* window with 52 containers, visits 446 nodes, and yields exactly ONE candidate
|
|
937
|
+
* that survives the caller's byte compare — the form itself.
|
|
928
938
|
*
|
|
929
|
-
* These are PROPOSALS only.
|
|
930
|
-
*
|
|
931
|
-
*
|
|
939
|
+
* These are PROPOSALS only. Every candidate still faces the byte-exact prefix
|
|
940
|
+
* compare and all three guards below, so a wrong proposal costs one bounded
|
|
941
|
+
* read and can never be voiced (exact-vs-approximate.md). */
|
|
932
942
|
export function formsOpenedBy(
|
|
933
943
|
ctx: MindContext,
|
|
934
944
|
query: Uint8Array,
|
package/src/mind/types.ts
CHANGED
|
@@ -65,12 +65,38 @@ export interface GraphSearchHost {
|
|
|
65
65
|
starts: ReadonlySet<number>;
|
|
66
66
|
};
|
|
67
67
|
chooseNext?(node: number): number | undefined;
|
|
68
|
+
/** The boundary positions of `bytes` under the engine's ONE boundary rule
|
|
69
|
+
* (geometry.ts's `contentBoundaries`), or undefined when the host has no
|
|
70
|
+
* space to ask. The join's key is an entity plus a prefix of the tail, and
|
|
71
|
+
* the prefix that names a stored relation ENDS on one of these boundaries —
|
|
72
|
+
* measured, 5 of 5 accepted keys over four join-firing queries, where the
|
|
73
|
+
* byte-by-byte scan spent 153 probes for 14 boundaries. Boundaries are
|
|
74
|
+
* content-defined and STABLE under prefix extension, which is why a corpus
|
|
75
|
+
* key's end is a boundary of the query's own fold of the same bytes. */
|
|
76
|
+
contentCuts?(bytes: Uint8Array): readonly number[];
|
|
68
77
|
/** The admission predicate — `traverse.ts`'s `leadsSomewhere`, its ONE
|
|
69
78
|
* definition: does this node bear an edge or a halo? Optional, so a bare
|
|
70
79
|
* host (a raw Store and nothing else) still works; when present, the search
|
|
71
80
|
* uses it rather than re-probing the store, which keeps the predicate
|
|
72
81
|
* single-defined AND memoised on the response-scoped struct cache. */
|
|
73
82
|
leadsSomewhere?(id: number): boolean;
|
|
83
|
+
/** Report a SEARCH REFUSAL into the rationale — the channel AGENTS §6
|
|
84
|
+
* requires: a callback threaded through a call chain must FEED the
|
|
85
|
+
* rationale, the way `GraphSearch`'s `onDerivation` feeds `traceDerivation`,
|
|
86
|
+
* never a channel of its own. Optional, so a bare host stays silent rather
|
|
87
|
+
* than crashing. */
|
|
88
|
+
reportSearch?(
|
|
89
|
+
name: string,
|
|
90
|
+
parts: ReadonlyArray<Uint8Array>,
|
|
91
|
+
note: string,
|
|
92
|
+
): void;
|
|
93
|
+
/** The CANONICAL resolver ({@link canonResolve}), optional like
|
|
94
|
+
* {@link leadsSomewhere}. The store's keys were written through the
|
|
95
|
+
* canonical fold, so a fact's `Gustaf Molander` and the deposited
|
|
96
|
+
* `gustaf molander` are the SAME node (measured inside a response: the
|
|
97
|
+
* canonical resolver maps the surface form to the deposited node while a raw
|
|
98
|
+
* resolve returns null). A bare host falls back to the plain probe. */
|
|
99
|
+
canonResolve?(bytes: Uint8Array): number | null;
|
|
74
100
|
}
|
|
75
101
|
|
|
76
102
|
// ═══════════════════════════════════════════════════════════════════════════
|
|
@@ -297,10 +323,10 @@ export type AItem =
|
|
|
297
323
|
export interface MindContext extends GraphSearchHost {
|
|
298
324
|
store: Store;
|
|
299
325
|
/** The work accumulator for the inference call in flight, or null when
|
|
300
|
-
*
|
|
301
|
-
*
|
|
302
|
-
*
|
|
303
|
-
*
|
|
326
|
+
* nothing is profiling — see src/meter.ts. WRITE-ONLY from the engine's point
|
|
327
|
+
* of view: no inference decision may read a counter, or determinism is gone
|
|
328
|
+
* (determinism.md). Every call site is `ctx.meter?.x++`, so an unprofiled
|
|
329
|
+
* response allocates nothing. */
|
|
304
330
|
meter: Meter | null;
|
|
305
331
|
space: Space;
|
|
306
332
|
alphabet: Alphabet;
|
package/src/store.ts
CHANGED
|
@@ -541,16 +541,16 @@ export interface Store {
|
|
|
541
541
|
// selected by identity hash so the choice is a property of each constituent
|
|
542
542
|
// and never of where it sits in the fold.
|
|
543
543
|
//
|
|
544
|
-
// DURABLE DERIVED STATE, NOT A CACHE.
|
|
544
|
+
// DURABLE DERIVED STATE, NOT A CACHE. caches.md permits a cache to cost only
|
|
545
545
|
// speed; this decides which terms enter a halo — a learned relation — so an
|
|
546
|
-
// eviction would change the geometry rather than slow it down.
|
|
546
|
+
// eviction would change the geometry rather than slow it down. It is
|
|
547
547
|
// therefore written like the canon index: computed once, kept, never
|
|
548
|
-
// budgeted.
|
|
549
|
-
//
|
|
550
|
-
//
|
|
551
|
-
//
|
|
552
|
-
//
|
|
553
|
-
//
|
|
548
|
+
// budgeted. Soundness rests on the set being INTRINSIC — minimality, `len ≥
|
|
549
|
+
// W` and non-domination are properties of the node's own subtree and do not
|
|
550
|
+
// move as the corpus grows. The one corpus-dependent reading, the hub
|
|
551
|
+
// exclusion, is deliberately NOT stored: it is applied by the caller at pour
|
|
552
|
+
// time over the ≤ k candidates, which is the drift companyProfile already
|
|
553
|
+
// documents as benign and one-directional.
|
|
554
554
|
//
|
|
555
555
|
// Backends that do not implement the pair leave both absent; companyProfile
|
|
556
556
|
// then recomputes the sketch per pour and simply loses the amortisation.
|
|
@@ -1789,18 +1789,18 @@ export abstract class AbstractStore implements Store {
|
|
|
1789
1789
|
* remainders must fit the budget. Scattered differences leave a wide
|
|
1790
1790
|
* middle and are rejected.
|
|
1791
1791
|
*
|
|
1792
|
-
* Every read here is CAPPED (
|
|
1793
|
-
*
|
|
1794
|
-
*
|
|
1795
|
-
*
|
|
1796
|
-
*
|
|
1797
|
-
*
|
|
1798
|
-
*
|
|
1799
|
-
*
|
|
1800
|
-
*
|
|
1801
|
-
*
|
|
1802
|
-
*
|
|
1803
|
-
*
|
|
1792
|
+
* Every read here is CAPPED (bounded-reads.md). It used to open with
|
|
1793
|
+
* `bytesPrefix(k, Number.MAX_SAFE_INTEGER)` — the ALL sentinel, i.e. the full
|
|
1794
|
+
* materialising `bytes()` read — on the deposit hot path, and only then
|
|
1795
|
+
* compare lengths. So a candidate the length test was about to reject had
|
|
1796
|
+
* already been reconstructed byte for byte. The LENGTHS decide first instead,
|
|
1797
|
+
* from the `contentLen` memo the interning order has already built bottom-up,
|
|
1798
|
+
* and the target's length is itself read under a cap: a target longer than
|
|
1799
|
+
* `la + W` is rejected without touching one of its bytes. Same semantics —
|
|
1800
|
+
* the old capped `b` read would have produced `a.length + W + 1` here and
|
|
1801
|
+
* failed the very same test — strictly fewer byte reads. The `+ 1` on each
|
|
1802
|
+
* byte cap keeps `_prefix`'s "complete reconstruction" test true, so the
|
|
1803
|
+
* results still cache. */
|
|
1804
1804
|
private differsByOneWindow(
|
|
1805
1805
|
kids: NodeId[],
|
|
1806
1806
|
targetId: NodeId,
|
package/test/08-storage.test.mjs
CHANGED
|
@@ -361,7 +361,7 @@ test("a halo accumulates poured signatures and gates on mass", async () => {
|
|
|
361
361
|
});
|
|
362
362
|
|
|
363
363
|
// A multi-turn conversation is deposited as ACCUMULATED-CONTEXT episodes — the
|
|
364
|
-
// pattern
|
|
364
|
+
// pattern example/train.ts uses:
|
|
365
365
|
// (t0) → t1
|
|
366
366
|
// (t0 + t1) → t2
|
|
367
367
|
// (t0 + t1 + t2) → t3
|
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
// 100-complete-grounding-trace.test.mjs — a declared-complete grounding SAYS
|
|
2
|
+
// SO in the rationale.
|
|
3
|
+
//
|
|
4
|
+
// THE GAP THIS CLOSES. `pipeline.ts` ends the derivation when the winning
|
|
5
|
+
// grounding carries `MechanismResult.complete` — the mechanism's own claim that
|
|
6
|
+
// the query IS a stored context, so its continuation is the whole read-out and
|
|
7
|
+
// a further pivot could only chain PAST the fact that produced the answer. The
|
|
8
|
+
// decision was correct and SILENT: no step, no note. A reader of the rationale
|
|
9
|
+
// therefore could not tell
|
|
10
|
+
//
|
|
11
|
+
// "the chain stopped because the query WAS the context" (complete)
|
|
12
|
+
//
|
|
13
|
+
// apart from
|
|
14
|
+
//
|
|
15
|
+
// "nothing followed" (no continuation)
|
|
16
|
+
//
|
|
17
|
+
// which are different claims about the same answer. AGENTS §6 makes that an
|
|
18
|
+
// instrumentation defect rather than a documentation gap: a bound that
|
|
19
|
+
// truncates must be reportable AT THE POINT it truncates, through the one
|
|
20
|
+
// surface that already exists.
|
|
21
|
+
//
|
|
22
|
+
// WHAT IS PINNED HERE.
|
|
23
|
+
// 1. the stop is REPORTED (step `completeGrounding`) when it happens;
|
|
24
|
+
// 2. the stop is REAL — the post-grounding extension did not run (no pivot or
|
|
25
|
+
// forward-absorb step), so the report cannot rot into a lie;
|
|
26
|
+
// 3. the report does NOT appear for an ordinary grounding, so it names a
|
|
27
|
+
// decision rather than decorating every response.
|
|
28
|
+
//
|
|
29
|
+
// The fixture is test/76's CARRIED frame: the continuation quotes its filler
|
|
30
|
+
// (`Run gcc <X>`), which is the shape the reference mechanism binds and
|
|
31
|
+
// declares complete.
|
|
32
|
+
|
|
33
|
+
import { test } from "node:test";
|
|
34
|
+
import assert from "node:assert/strict";
|
|
35
|
+
import { Mind } from "../dist/src/index.js";
|
|
36
|
+
import { SQliteStore } from "../dist/src/store-sqlite.js";
|
|
37
|
+
|
|
38
|
+
const CARRIED = [
|
|
39
|
+
["How do I compile hello.c?", "Run gcc hello.c"],
|
|
40
|
+
["How do I compile server.c?", "Run gcc server.c"],
|
|
41
|
+
["How do I compile parser.c?", "Run gcc parser.c"],
|
|
42
|
+
];
|
|
43
|
+
|
|
44
|
+
/** The frame fixture — the same one test/76 pins the binding on. */
|
|
45
|
+
async function frame() {
|
|
46
|
+
const m = new Mind({ seed: 7, store: new SQliteStore({ path: ":memory:" }) });
|
|
47
|
+
await m.ingest(CARRIED);
|
|
48
|
+
return m;
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/** A grounding that is NOT declared complete: one plain learnt edge. */
|
|
52
|
+
async function plainChain() {
|
|
53
|
+
const m = new Mind({ seed: 7, store: new SQliteStore({ path: ":memory:" }) });
|
|
54
|
+
await m.ingest([["Eva director", "The director of Eva is Gustaf Molander."]]);
|
|
55
|
+
return m;
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
const moves = (steps) => steps.map((s) => s.mechanism.at(-1));
|
|
59
|
+
|
|
60
|
+
test("a binding that declares itself complete reports the stop", async () => {
|
|
61
|
+
const m = await frame();
|
|
62
|
+
const steps = [];
|
|
63
|
+
await m.respondText("How do I compile main.c?", (s) => steps.push(s));
|
|
64
|
+
|
|
65
|
+
const stop = steps.find((s) => s.mechanism.at(-1) === "completeGrounding");
|
|
66
|
+
assert.ok(stop, "the rationale must report the declared-complete stop");
|
|
67
|
+
assert.match(
|
|
68
|
+
stop.note,
|
|
69
|
+
/declared complete/,
|
|
70
|
+
"the note must carry the CLAIM, not a description of the answer",
|
|
71
|
+
);
|
|
72
|
+
});
|
|
73
|
+
|
|
74
|
+
test("the reported stop is real: the extension did not run", async () => {
|
|
75
|
+
const m = await frame();
|
|
76
|
+
const steps = [];
|
|
77
|
+
const answer = await m.respondText(
|
|
78
|
+
"How do I compile main.c?",
|
|
79
|
+
(s) => steps.push(s),
|
|
80
|
+
);
|
|
81
|
+
|
|
82
|
+
// The binding spliced the asker's referent (behaviour under test, asserted
|
|
83
|
+
// WITHOUT a trace below — a trace can change an answer, so the step and the
|
|
84
|
+
// bytes are pinned separately).
|
|
85
|
+
assert.ok(answer.length > 0, "the binding must answer");
|
|
86
|
+
const seen = moves(steps);
|
|
87
|
+
assert.equal(
|
|
88
|
+
seen.includes("pivotStep") || seen.includes("absorbForward"),
|
|
89
|
+
false,
|
|
90
|
+
"a complete grounding must not be extended",
|
|
91
|
+
);
|
|
92
|
+
});
|
|
93
|
+
|
|
94
|
+
test("the non-traced answer is the spliced one", async () => {
|
|
95
|
+
const m = await frame();
|
|
96
|
+
const answer = await m.respondText("How do I compile main.c?");
|
|
97
|
+
assert.equal(answer.replace(/\0+/g, "").trim(), "Run gcc main.c");
|
|
98
|
+
});
|
|
99
|
+
|
|
100
|
+
test("an ordinary grounding reports no complete-grounding stop", async () => {
|
|
101
|
+
const m = await plainChain();
|
|
102
|
+
const steps = [];
|
|
103
|
+
await m.respondText("Eva director", (s) => steps.push(s));
|
|
104
|
+
assert.equal(
|
|
105
|
+
moves(steps).includes("completeGrounding"),
|
|
106
|
+
false,
|
|
107
|
+
"the step names a decision, not every response",
|
|
108
|
+
);
|
|
109
|
+
});
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
// 101-alignment-gap-bound.test.mjs — the alignment's gap bound SAYS SO when it
|
|
2
|
+
// bites.
|
|
3
|
+
//
|
|
4
|
+
// THE GAP THIS CLOSES. `alignAround` matches a query against a stored context
|
|
5
|
+
// by sweeping for the next common run of ≥ W bytes, with each side's gap
|
|
6
|
+
// bounded by `chainReach(W)` = W². When the sweep ends with material still
|
|
7
|
+
// unmatched on BOTH sides, the alignment stopped because of that BOUND — and
|
|
8
|
+
// nothing reported it: `match.ts` emitted no trace at all. A reader of the
|
|
9
|
+
// rationale therefore saw a substituted answer (the frame filled with ANOTHER
|
|
10
|
+
// instance's filler) with no way to tell that the answer's own span had been
|
|
11
|
+
// refused by a cap.
|
|
12
|
+
//
|
|
13
|
+
// MEASURED BOUNDARY (W = 4, cap = 16). Against a learned
|
|
14
|
+
// `Book a table at <filler> tonight.` → `Your table at <filler> is booked.`
|
|
15
|
+
// frame, a NOVEL filler binds while it fits the cap and is replaced by a stored
|
|
16
|
+
// instance's filler once it does not:
|
|
17
|
+
//
|
|
18
|
+
// filler 7 B → answer quotes the asker's filler (bound)
|
|
19
|
+
// filler 12 B → answer quotes the asker's filler (bound)
|
|
20
|
+
// filler 18 B → answer carries ANOTHER filler (substituted)
|
|
21
|
+
// filler 30 B → answer carries ANOTHER filler (substituted)
|
|
22
|
+
//
|
|
23
|
+
// THIS FILE PINS THE REPORT, not the substitution: the step must exist and must
|
|
24
|
+
// carry the derived cap, so the bound is auditable at the point it truncates
|
|
25
|
+
// (AGENTS §6). The substitution itself is the scope defect a later lot fixes;
|
|
26
|
+
// when it is fixed, this file's behavioural control (a filler inside the cap is
|
|
27
|
+
// quoted back) stays true and the report stays reachable.
|
|
28
|
+
|
|
29
|
+
import { test } from "node:test";
|
|
30
|
+
import assert from "node:assert/strict";
|
|
31
|
+
import { Mind } from "../dist/src/index.js";
|
|
32
|
+
import { alignAround } from "../dist/src/mind/match.js";
|
|
33
|
+
import { SQliteStore } from "../dist/src/store-sqlite.js";
|
|
34
|
+
|
|
35
|
+
const WORDS =
|
|
36
|
+
("alpha bravo charlie delta echo foxtrot golf hotel india juliet kilo lima " +
|
|
37
|
+
"mike november oscar papa quebec romeo sierra tango uniform victor whiskey " +
|
|
38
|
+
"xray yankee zulu amber bronze copper dahlia ember fjord gossamer harbour " +
|
|
39
|
+
"indigo jasmine kestrel lantern marigold nectar opal pewter quartz ripple " +
|
|
40
|
+
"saffron thistle umber violet willow xenon yarrow").split(" ");
|
|
41
|
+
|
|
42
|
+
const short = (i) => `${WORDS[i % 50]}${i}`;
|
|
43
|
+
const long = (i) =>
|
|
44
|
+
`${WORDS[i % 50]} ${WORDS[(i * 3) % 50]} ${WORDS[(i * 7) % 50]} street ${i}`;
|
|
45
|
+
|
|
46
|
+
/** The frame whose continuation quotes its filler, with mixed filler widths. */
|
|
47
|
+
async function frame(n = 60, opts = {}) {
|
|
48
|
+
const store = new SQliteStore({ path: ":memory:", D: 1024 });
|
|
49
|
+
const mind = new Mind({ seed: 7, store, ...opts });
|
|
50
|
+
const pairs = [];
|
|
51
|
+
for (let i = 0; i < n; i++) {
|
|
52
|
+
const f = i % 3 === 0 ? long(i) : short(i);
|
|
53
|
+
pairs.push([
|
|
54
|
+
`Book a table at ${f} tonight.`,
|
|
55
|
+
`Your table at ${f} is booked.`,
|
|
56
|
+
]);
|
|
57
|
+
}
|
|
58
|
+
await mind.ingest(pairs);
|
|
59
|
+
return { store, mind };
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
/** A novel filler of exactly `len` bytes, never ingested. */
|
|
63
|
+
const novel = (len) => "zephyr quartz lantern ".repeat(3).slice(0, len).trim();
|
|
64
|
+
|
|
65
|
+
// THE LAW'S ASSERTION, straight on the aligner. There is no bound to report
|
|
66
|
+
// any more: the sweep's work is proportional to the bytes a run spans, so a
|
|
67
|
+
// divergence far past the old arity bound (chainReach(W) = 16) is simply
|
|
68
|
+
// bridged, on both sides, by finding the next common run.
|
|
69
|
+
test("a divergence far past the old arity bound is bridged, not truncated", () => {
|
|
70
|
+
const enc = new TextEncoder();
|
|
71
|
+
const head = "the common head of the frame ";
|
|
72
|
+
const tail = " and the common tail of the frame";
|
|
73
|
+
const q = enc.encode(head + "a".repeat(60) + tail);
|
|
74
|
+
const c = enc.encode(head + "b".repeat(60) + tail);
|
|
75
|
+
const at = head.length - 1; // the seed: the shared head's own boundary
|
|
76
|
+
const { matched, gaps } = alignAround(
|
|
77
|
+
{ space: { maxGroup: 4 } },
|
|
78
|
+
q,
|
|
79
|
+
c,
|
|
80
|
+
at,
|
|
81
|
+
at,
|
|
82
|
+
);
|
|
83
|
+
assert.equal(
|
|
84
|
+
matched.length,
|
|
85
|
+
2,
|
|
86
|
+
"both common runs must be found: " + JSON.stringify(matched),
|
|
87
|
+
);
|
|
88
|
+
assert.equal(gaps.length, 1, "with exactly one substitution between them");
|
|
89
|
+
const g = gaps[0];
|
|
90
|
+
assert.ok(
|
|
91
|
+
g.qe - g.qs >= 60 && g.ce - g.cs >= 60,
|
|
92
|
+
"the substitution's extent is the pair's own: " + JSON.stringify(g),
|
|
93
|
+
);
|
|
94
|
+
});
|
|
95
|
+
|
|
96
|
+
test("control: a filler inside the cap is quoted back, and the fixture is live", async () => {
|
|
97
|
+
const filler = novel(12);
|
|
98
|
+
const { mind } = await frame();
|
|
99
|
+
const answer = (await mind.respondText(
|
|
100
|
+
`Book a table at ${filler} tonight.`,
|
|
101
|
+
)).replace(/\0+/g, "").trim();
|
|
102
|
+
assert.ok(
|
|
103
|
+
answer.includes(filler),
|
|
104
|
+
"inside the cap the frame binds the asker's own filler",
|
|
105
|
+
);
|
|
106
|
+
});
|