@hviana/sema 0.8.0 → 0.8.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (119) hide show
  1. package/AGENTS.md +22 -1
  2. package/DATASETS.md +1 -1
  3. package/dist/example/train_base/config.js +2 -2
  4. package/dist/example/train_base/corpora/massive.js +1 -1
  5. package/dist/example/train_base/readers.js +1 -1
  6. package/dist/src/config.d.ts +17 -0
  7. package/dist/src/config.js +18 -0
  8. package/dist/src/geometry.d.ts +10 -10
  9. package/dist/src/geometry.js +25 -24
  10. package/dist/src/meter.d.ts +29 -12
  11. package/dist/src/meter.js +58 -14
  12. package/dist/src/mind/attention.js +12 -12
  13. package/dist/src/mind/bridge.d.ts +8 -8
  14. package/dist/src/mind/bridge.js +33 -32
  15. package/dist/src/mind/corpus.d.ts +40 -0
  16. package/dist/src/mind/corpus.js +149 -0
  17. package/dist/src/mind/graph-search.d.ts +7 -8
  18. package/dist/src/mind/graph-search.js +244 -32
  19. package/dist/src/mind/index.d.ts +3 -1
  20. package/dist/src/mind/index.js +1 -0
  21. package/dist/src/mind/junction.d.ts +1 -1
  22. package/dist/src/mind/junction.js +8 -8
  23. package/dist/src/mind/learning.js +36 -35
  24. package/dist/src/mind/match.d.ts +8 -3
  25. package/dist/src/mind/match.js +156 -71
  26. package/dist/src/mind/mechanisms/cast.js +18 -2
  27. package/dist/src/mind/mechanisms/cover.js +19 -12
  28. package/dist/src/mind/mechanisms/prefix-completion.js +24 -24
  29. package/dist/src/mind/mechanisms/recall.js +38 -40
  30. package/dist/src/mind/mechanisms/reference.js +16 -16
  31. package/dist/src/mind/mind.d.ts +61 -7
  32. package/dist/src/mind/mind.js +72 -2
  33. package/dist/src/mind/pipeline-mechanism.d.ts +10 -8
  34. package/dist/src/mind/pipeline-mechanism.js +25 -21
  35. package/dist/src/mind/pipeline.d.ts +9 -9
  36. package/dist/src/mind/pipeline.js +49 -29
  37. package/dist/src/mind/primitives.d.ts +5 -5
  38. package/dist/src/mind/primitives.js +5 -5
  39. package/dist/src/mind/reasoning.d.ts +5 -1
  40. package/dist/src/mind/reasoning.js +54 -1
  41. package/dist/src/mind/recognition.d.ts +14 -13
  42. package/dist/src/mind/recognition.js +23 -23
  43. package/dist/src/mind/resonance.js +21 -21
  44. package/dist/src/mind/traverse.d.ts +54 -52
  45. package/dist/src/mind/traverse.js +83 -73
  46. package/dist/src/mind/types.d.ts +26 -4
  47. package/dist/src/store.d.ts +12 -12
  48. package/dist/src/store.js +12 -12
  49. package/docs/INDEX.md +2 -2
  50. package/docs/architecture/exact-vs-approximate.md +2 -1
  51. package/docs/architecture/fold-contract.md +1 -1
  52. package/docs/failures/tempting-but-wrong.md +33 -5
  53. package/docs/harness/gates.md +7 -7
  54. package/example/train_base/config.ts +2 -2
  55. package/example/train_base/corpora/massive.ts +1 -1
  56. package/example/train_base/readers.ts +1 -1
  57. package/jsr.json +1 -1
  58. package/package.json +1 -1
  59. package/src/config.ts +35 -0
  60. package/src/geometry.ts +25 -24
  61. package/src/meter.ts +61 -14
  62. package/src/mind/attention.ts +12 -12
  63. package/src/mind/bridge.ts +33 -32
  64. package/src/mind/corpus.ts +202 -0
  65. package/src/mind/graph-search.ts +261 -31
  66. package/src/mind/index.ts +8 -1
  67. package/src/mind/junction.ts +8 -8
  68. package/src/mind/learning.ts +36 -35
  69. package/src/mind/match.ts +163 -73
  70. package/src/mind/mechanisms/cast.ts +17 -1
  71. package/src/mind/mechanisms/cover.ts +18 -12
  72. package/src/mind/mechanisms/prefix-completion.ts +24 -24
  73. package/src/mind/mechanisms/recall.ts +38 -40
  74. package/src/mind/mechanisms/reference.ts +16 -16
  75. package/src/mind/mind.ts +129 -7
  76. package/src/mind/pipeline-mechanism.ts +25 -21
  77. package/src/mind/pipeline.ts +63 -38
  78. package/src/mind/primitives.ts +5 -5
  79. package/src/mind/reasoning.ts +55 -0
  80. package/src/mind/recognition.ts +23 -23
  81. package/src/mind/resonance.ts +21 -21
  82. package/src/mind/traverse.ts +83 -73
  83. package/src/mind/types.ts +30 -4
  84. package/src/store.ts +20 -20
  85. package/test/08-storage.test.mjs +1 -1
  86. package/test/100-complete-grounding-trace.test.mjs +109 -0
  87. package/test/101-alignment-gap-bound.test.mjs +106 -0
  88. package/test/102-production-composes-at-scale.test.mjs +110 -0
  89. package/test/103-alignment-gap-budget.test.mjs +89 -0
  90. package/test/104-composition-is-reported.test.mjs +90 -0
  91. package/test/105-derive-through-reports-its-refusal.test.mjs +113 -0
  92. package/test/106-the-join-fires.test.mjs +94 -0
  93. package/test/107-the-join-is-counted.test.mjs +81 -0
  94. package/test/108-the-join-chains.test.mjs +78 -0
  95. package/test/109-the-pivot-is-counted.test.mjs +60 -0
  96. package/test/110-the-reasoner-stops-when-the-question-is-answered.test.mjs +91 -0
  97. package/test/111-the-cover-assembly-is-counted.test.mjs +74 -0
  98. package/test/112-the-exploration-does-not-grow-with-the-hub.test.mjs +89 -0
  99. package/test/113-the-rationale-payload-is-bounded.test.mjs +84 -0
  100. package/test/114-alignment-budget-is-per-sweep.test.mjs +93 -0
  101. package/test/116-the-extension-is-gated-by-the-pipelines-own-remainder.test.mjs +100 -0
  102. package/test/117-corpus-search.test.mjs +171 -0
  103. package/test/14-scaling.test.mjs +10 -7
  104. package/test/35-prefix-edge.test.mjs +1 -1
  105. package/test/40-choosenext-scale-guard.test.mjs +16 -17
  106. package/test/56-bridge-identity-admission.test.mjs +6 -6
  107. package/test/70-prefix-completion.test.mjs +4 -3
  108. package/test/72-prefix-candidate-supply.test.mjs +3 -3
  109. package/test/73-scaffolding-only-bridge-abstains.test.mjs +6 -6
  110. package/test/75-multiturn-context-optimisation.test.mjs +5 -5
  111. package/test/76-reference-binding.test.mjs +6 -1
  112. package/test/84-composed-answer-honesty.test.mjs +5 -6
  113. package/test/88-dependency-footprint.test.mjs +1 -1
  114. package/test/89-completion-recursion.test.mjs +47 -19
  115. package/test/90-connector-read-cap.test.mjs +10 -8
  116. package/test/93-regime-prediction.test.mjs +10 -10
  117. package/test/94-cross-region-budget.test.mjs +2 -2
  118. package/test/95-wide-resonance-removed.test.mjs +8 -7
  119. package/test/96-bytes-walk-termination.test.mjs +3 -3
@@ -380,15 +380,15 @@ export async function pivotInto(
380
380
  // Byte containment, longest wins — the answer literally contains the
381
381
  // pivot's bytes, and the biggest well-evidenced span is the real pivot.
382
382
  //
383
- // REAL SATURATION, not a hard cap: the score IS the candidate's byte
384
- // length, so the scan is DECIDED the moment the first candidate that passes
385
- // every filter is found in DESCENDING length order — a shorter candidate can
386
- // never outscore it. `contentLen` (the prefix-capped length read, §2.8) is
387
- // the cheap ordering key, and the first-inserted tie-break is made explicit
388
- // (`a.index - b.index`) so equal lengths keep `scored`'s insertion order —
389
- // exactly the tie argmaxBy(strict) used to keep. The bytes of at most ONE
390
- // winning candidate are read; every shorter candidate the probes proposed is
391
- // skipped without reconstruction, where the old argmax read them all.
383
+ // REAL SATURATION, not a hard cap: the score IS the candidate's byte length,
384
+ // so the scan is DECIDED the moment the first candidate that passes every
385
+ // filter is found in DESCENDING length order — a shorter candidate can never
386
+ // outscore it. `contentLen` (the prefix-capped length read, bounded-reads.md)
387
+ // is the cheap ordering key, and the first-inserted tie-break is made
388
+ // explicit (`a.index - b.index`) so equal lengths keep `scored`'s insertion
389
+ // order — exactly the tie argmaxBy(strict) used to keep. The bytes of at most
390
+ // ONE winning candidate are read; every shorter candidate the probes proposed
391
+ // is skipped without reconstruction, where the old argmax read them all.
392
392
  const ranked = [...scored.keys()]
393
393
  .map((id, index) => ({
394
394
  id,
@@ -399,11 +399,11 @@ export async function pivotInto(
399
399
  let pivotId: number | null = null;
400
400
  for (const c of ranked) {
401
401
  const id = c.id;
402
- // A ZERO-LENGTH candidate is not a pivot. `argmaxBy(…, 0, strict)` used to
402
+ // A ZERO-LENGTH candidate is not a pivot. `argmaxBy(…, 0, strict)` used to
403
403
  // carry this floor in its threshold argument, and dropping it here would
404
404
  // admit an empty node: `indexOf(answer, <empty>)` returns 0, so every
405
- // filter below passes and the chain would hop through nothing (§2.13 —
406
- // empty bytes are truthy).
405
+ // filter below passes and the chain would hop through nothing
406
+ // (INVARIANTS.md — empty bytes are truthy).
407
407
  if (c.len === 0) continue;
408
408
  // A PIVOT MUST BE A THING THE CORPUS DEPOSITED, NOT A PIECE OF ONE.
409
409
  // "Longest wins" ranks candidates but never asks whether the winner is
@@ -439,15 +439,15 @@ export async function pivotInto(
439
439
  // a span that was never a fact on its own is not one to step through.
440
440
  // No constant enters — it is a structural predicate, not a threshold.
441
441
  if (ctx.store.hasParents(id) || ctx.store.hasContainers(id)) continue;
442
- // A candidate whose bytes are LONGER than the answer cannot be a
443
- // substring of it — `indexOf` would return −1 regardless. Prune by
444
- // length BEFORE reconstructing the bytes: `read` is an UNCAPPED read
445
- // (AGENTS §2.8), and a resonated context far longer than the answer is
446
- // exactly the candidate that makes it cost a whole deposit's worth of
447
- // reconstruction for a containment test that must fail. `contentLen`
448
- // with the `answer.length + 1` cap is the prefix-capped length read the
449
- // same contract prescribes; the prune is byte-identical to the old
450
- // `indexOf` miss (it returns −1 for a needle longer than the haystack).
442
+ // A candidate whose bytes are LONGER than the answer cannot be a substring
443
+ // of it — `indexOf` would return −1 regardless. Prune by length BEFORE
444
+ // reconstructing the bytes: `read` is an UNCAPPED read (bounded-reads.md),
445
+ // and a resonated context far longer than the answer is exactly the
446
+ // candidate that makes it cost a whole deposit's worth of reconstruction
447
+ // for a containment test that must fail. `contentLen` with the
448
+ // `answer.length + 1` cap is the prefix-capped length read the same
449
+ // contract prescribes; the prune is byte-identical to the old `indexOf`
450
+ // miss (it returns −1 for a needle longer than the haystack).
451
451
  if (c.len > answer.length) continue;
452
452
  const bytes = read(ctx, id);
453
453
  if (indexOf(answer, bytes, 0) < 0) continue;
@@ -32,11 +32,11 @@ interface StructCache {
32
32
  hasParents: Map<number, boolean>;
33
33
  }
34
34
  //
35
- // Budgeted on the same terms as the reach memo below (AGENTS §2.12): these
36
- // three maps are cleared on every write, but a long read-only session over a
37
- // large store converges on one entry per node per map with nothing to bound
38
- // it. Past the cap all three are dropped together and re-derived, costing
39
- // cold structural probes and never a wrong answer.
35
+ // Budgeted on the same terms as the reach memo below (caches.md): these three
36
+ // maps are cleared on every write, but a long read-only session over a large
37
+ // store converges on one entry per node per map with nothing to bound it. Past
38
+ // the cap all three are dropped together and re-derived, costing cold
39
+ // structural probes and never a wrong answer.
40
40
  const STRUCT_MEMO_MAX = 100_000;
41
41
  const structCaches = new WeakMap<object, StructCache>();
42
42
 
@@ -55,21 +55,22 @@ const structCaches = new WeakMap<object, StructCache>();
55
55
  // battery repeatedly reaches the same corpus scaffolding even when its
56
56
  // surface questions differ.
57
57
  //
58
- // Budgeted, not unbounded (AGENTS §2.12): past the cap the whole map is
59
- // dropped and re-derived, costing a cold climb and never a wrong answer.
58
+ // Budgeted, not unbounded (caches.md): past the cap the whole map is dropped
59
+ // and
60
+ // re-derived, costing a cold climb and never a wrong answer.
60
61
  const REACH_MEMO_MAX = 100_000;
61
62
  const reachCaches = new WeakMap<object, Map<number, AncestorReach>>();
62
63
 
63
64
  /** The reach memo this ask should use — see the note above.
64
65
  *
65
- * A TRACED response always gets a fresh, empty one. `AncestorReach`'s
66
- * `visited`/`maxDepth`/`saturation` fields are populated only when a trace
67
- * is attached, so an entry deposited by an untraced earlier turn would
68
- * silently black out the reach detail of a later traced one; and the trace's
69
- * reach payload is serialised by ITERATING this map, which must therefore
70
- * hold what THIS climb consulted, not the whole conversation's history.
71
- * Consistent with AGENTS §2.11: a traced response is a different machine —
72
- * never benchmark with a trace attached. */
66
+ * A TRACED response always gets a fresh, empty one. `AncestorReach`'s
67
+ * `visited`/`maxDepth`/`saturation` fields are populated only when a trace is
68
+ * attached, so an entry deposited by an untraced earlier turn would silently
69
+ * black out the reach detail of a later traced one; and the trace's reach
70
+ * payload is serialised by ITERATING this map, which must therefore hold what
71
+ * THIS climb consulted, not the whole conversation's history. Consistent with
72
+ * memoization.md: a traced response is a different machine — never benchmark
73
+ * with a trace attached. */
73
74
  export function sharedReachMemo(
74
75
  ctx: MindContext,
75
76
  ): Map<number, AncestorReach> {
@@ -481,13 +482,14 @@ export function bearsEdge(ctx: MindContext, id: number): boolean {
481
482
  }
482
483
 
483
484
  /** Whether a node LEADS SOMEWHERE — it bears a continuation edge or a halo.
484
- * The admission predicate recognition filters sites with (HOW_IT_WORKS
485
- * §15.3): a form that leads nowhere contributes nothing to any derivation.
486
- * Runs once per candidate span on the recognition hot path — `hasNext` is
487
- * cached per response (the same flat-branch ids are probed across prefix
488
- * variants by canonicalChunkId). `hasHalo` is not cached: it's a single
489
- * indexed point probe per candidate, and the candidates that reach this
490
- * check have already been filtered by hasNext above in edgeAncestors. */
485
+ * The admission predicate recognition filters sites with (cover.md): a form
486
+ * that
487
+ * leads nowhere contributes nothing to any derivation. Runs once per candidate
488
+ * span on the recognition hot path — `hasNext` is cached per response (the same
489
+ * flat-branch ids are probed across prefix variants by canonicalChunkId).
490
+ * `hasHalo` is not cached: it's a single indexed point probe per candidate, and
491
+ * the candidates that reach this check have already been filtered by hasNext
492
+ * above in edgeAncestors. */
491
493
  export function leadsSomewhere(ctx: MindContext, id: number): boolean {
492
494
  const memo = getStructCache(ctx);
493
495
  if (cachedHasNext(ctx, id, memo)) return true;
@@ -539,10 +541,10 @@ function boundFor(contextCount: number): number {
539
541
  }
540
542
 
541
543
  /** Cap a candidate list at the hub bound √N (insertion order) — the ONE
542
- * fan-out convention every walk and disambiguation uses (see HOW_IT_WORKS
543
- * §8.6). A node connected to more than √N others is a hub whose individual
544
- * connections carry ~no discriminative information; materialising or scoring
545
- * them all would make single decisions scale with the corpus. */
544
+ * fan-out convention every walk and disambiguation uses (see bounded-reads.md).
545
+ * A node connected to more than √N others is a hub whose individual connections
546
+ * carry ~no discriminative information; materialising or scoring them all would
547
+ * make single decisions scale with the corpus. */
546
548
  export function hubCap<T>(
547
549
  ctx: MindContext,
548
550
  ids: readonly T[],
@@ -581,16 +583,16 @@ export function contains(
581
583
  * the EXACT half's veto on calling them synonyms.
582
584
  *
583
585
  * Halos measure company, and the strongest company any two forms can keep is
584
- * standing next to each other: a question and its answer co-occur in every
585
- * episode that taught the pair, so their halos SHOULD be similar, and on a
586
- * conversational store they are (measured on the CONV fixture: consecutive
587
- * turns at 0.809 against a 0.516 concept threshold). A gate reading halo
588
- * cosine alone therefore reads adjacency as synonymy and revoices an answer
589
- * in the words of the question it answers — "it hangs in madrid" spliced back
590
- * into "where is it kept now". The distributional layer cannot tell the two
591
- * relations apart, because to it they are the same observation; the exact
592
- * half can, for free, because it stored the edge. §4.1's division of labour
593
- * exactly: approximate proposes, exact decides.
586
+ * standing next to each other: a question and its answer co-occur in every
587
+ * episode that taught the pair, so their halos SHOULD be similar, and on a
588
+ * conversational store they are (measured on the CONV fixture: consecutive
589
+ * turns at 0.809 against a 0.516 concept threshold). A gate reading halo cosine
590
+ * alone therefore reads adjacency as synonymy and revoices an answer in the
591
+ * words of the question it answers — "it hangs in madrid" spliced back into
592
+ * "where is it kept now". The distributional layer cannot tell the two
593
+ * relations apart, because to it they are the same observation; the exact half
594
+ * can, for free, because it stored the edge. halo-sketch.md's division of
595
+ * labour exactly: approximate proposes, exact decides.
594
596
  *
595
597
  * Read LIMITed in both directions at the hub bound — a common continuation's
596
598
  * fan-in is corpus-sized, and no single decision may scale with it. */
@@ -745,26 +747,33 @@ export function chooseNext(
745
747
  // NO consensusFloor gate here (tried and reverted — see
746
748
  // test/40-choosenext-scale-guard.test.mjs): that floor is calibrated for
747
749
  // POOLED, IDF-weighted CLIMB VOTES (recallByResonance, commitVotes), where
748
- // each corroborating region contributes at most ln N and the floor grows
749
- // with N exactly as that per-region ceiling does (HOW_IT_WORKS.md §8.6).
750
- // `bestSupport` here is a different kind of quantity — a raw prevCount of
751
- // how many training contexts predicted ONE destination, bounded by how
752
- // often that specific fact was retold, never by corpus size N. Gating an
753
- // N-invariant count against an N-growing threshold guarantees failure
754
- // once N is large enough, discarding genuinely, structurally dominant
755
- // edges (observed: a fact corroborated 2-to-1-1-1 refused at N≈325K,
756
- // falling back to a noisy concept-hop). The loop above already IS the
757
- // "genuinely competing" test: a tie leaves first-inserted as the pick
758
- // (test/30's own pinned behaviour); a strict winner is real evidence
759
- // regardless of corpus scale. Matches HOW_IT_WORKS.md §25's own
760
- // chooseNext pseudocode, which has no such floor.
750
+ // each corroborating region contributes at most ln N and the floor grows with
751
+ // N exactly as that per-region ceiling does (thresholds.md). `bestSupport`
752
+ // here is a different kind of quantity — a raw prevCount of how many training
753
+ // contexts predicted ONE destination, bounded by how often that specific fact
754
+ // was retold, never by corpus size N. Gating an N-invariant count against an
755
+ // N-growing threshold guarantees failure once N is large enough, discarding
756
+ // genuinely, structurally dominant edges (observed: a fact corroborated
757
+ // 2-to-1-1-1 refused at N≈325K, falling back to a noisy concept-hop). The
758
+ // loop above already IS the "genuinely competing" test: a tie leaves
759
+ // first-inserted as the pick (test/30's own pinned behaviour); a strict
760
+ // winner is real evidence regardless of corpus scale. Matches `chooseNext`'s
761
+ // own pseudocode, which has no such floor.
761
762
 
762
763
  // Trace is built lazily — the filter + map below only execute when a
763
764
  // trace listener is attached, so the common (no-trace) path pays only
764
765
  // for the prevCount calls in the loop above, never for extra rItemShort
765
766
  // byte-reads.
766
767
  if (ctx.trace) {
767
- const others = capped.filter((c) => c !== best);
768
+ // A BOUNDED SAMPLE, AND THE COUNT. The step used to carry EVERY candidate
769
+ // it weighed — measured on the trained store, 1559 out-items in one step
770
+ // (hubBound's own size) and 1082 in another (the hub's degree). The
771
+ // rationale's job is to explain the CHOICE, and the count is what says how
772
+ // wide the field was; the declared candidate budget (`recallQueryK`) is what
773
+ // bounds the sample, so no number is invented here.
774
+ const others = capped
775
+ .filter((c) => c !== best)
776
+ .slice(0, ctx.cfg.rationaleSampleK);
768
777
  ctx.trace.step(
769
778
  "disambiguate",
770
779
  [rItemShort(ctx, best, "halo-evidence", bestSupport)],
@@ -838,12 +847,13 @@ function rItemShort(
838
847
  * W-window it spells is contained by more places than the hub bound allows,
839
848
  * i.e. the whole query is corpus-global scaffolding.
840
849
  *
841
- * WHAT IT IS FOR. Several mechanisms ground a query through the literal
842
- * spans it did NOT explain, and those spans are the whole of their evidence.
843
- * When every one of them is a hub, the query says nothing the corpus can be
844
- * held to, and grounding it means picking one of thousands of continuations
845
- * it gives no evidence for — a fabrication whatever the answer happens to be.
846
- * Answering with silence there is the honest degradation contract (§2.13).
850
+ * WHAT IT IS FOR. Several mechanisms ground a query through the literal spans
851
+ * it
852
+ * did NOT explain, and those spans are the whole of their evidence. When every
853
+ * one of them is a hub, the query says nothing the corpus can be held to, and
854
+ * grounding it means picking one of thousands of continuations it gives no
855
+ * evidence for — a fabrication whatever the answer happens to be. Answering
856
+ * with silence there is the honest degradation contract (INVARIANTS.md).
847
857
  *
848
858
  * MEASURED SEPARATION (trained store, hubBound 571) — this is categorical,
849
859
  * not marginal, and it is why the predicate lives here rather than being
@@ -860,11 +870,11 @@ function rItemShort(
860
870
  * evidence and sit on the SAME side as the correct ones, so this predicate
861
871
  * is not what makes them silent and cannot be credited for them.
862
872
  *
863
- * NO NEW THRESHOLD (§2.2): `hubBound` is the √N reading of "hub" used
864
- * everywhere, and the containment read is clamped to it exactly as every
865
- * other fan-out read is (§2.8). A query with no stored window at all is NOT
866
- * scaffolding-only — it has no evidence either way, and its callers already
867
- * refuse it on their own terms. */
873
+ * NO NEW THRESHOLD (thresholds.md): `hubBound` is the √N reading of "hub" used
874
+ * everywhere, and the containment read is clamped to it exactly as every other
875
+ * fan-out read is (bounded-reads.md). A query with no stored window at all is
876
+ * NOT scaffolding-only — it has no evidence either way, and its callers already
877
+ * refuse it on their own terms. */
868
878
  export function allWindowsAreScaffolding(
869
879
  ctx: MindContext,
870
880
  query: Uint8Array,
@@ -916,19 +926,19 @@ export function allWindowsAreScaffolding(
916
926
  * by climbing containment then parents. Nothing is added to the write side;
917
927
  * this reads an index training already built.
918
928
  *
919
- * BOUNDED (§2.8), AND WITH NO NEW THRESHOLD. The window whose containment is
920
- * SMALLEST carries the most evidence, and one saturated at `hubBound` carries
921
- * none — that is the same √N reading of "hub" the rest of the mind uses, not
922
- * a tuned knob. The upward walk spends a budget of `hubBound` nodes and
923
- * fans out by W, so a hub query enumerates nothing and the caller stays
924
- * silent rather than guessing (§2.13). Measured on the trained store: the
925
- * photosynthesis form at a one-byte truncation picks a window with 52
926
- * containers, visits 446 nodes, and yields exactly ONE candidate that
927
- * survives the caller's byte compare — the form itself.
929
+ * BOUNDED (bounded-reads.md), AND WITH NO NEW THRESHOLD. The window whose
930
+ * containment is SMALLEST carries the most evidence, and one saturated at
931
+ * `hubBound` carries none — that is the same √N reading of "hub" the rest of
932
+ * the mind uses, not a tuned knob. The upward walk spends a budget of
933
+ * `hubBound` nodes and fans out by W, so a hub query enumerates nothing and the
934
+ * caller stays silent rather than guessing (INVARIANTS.md). Measured on the
935
+ * trained store: the photosynthesis form at a one-byte truncation picks a
936
+ * window with 52 containers, visits 446 nodes, and yields exactly ONE candidate
937
+ * that survives the caller's byte compare — the form itself.
928
938
  *
929
- * These are PROPOSALS only. Every candidate still faces the byte-exact
930
- * prefix compare and all three guards below, so a wrong proposal costs one
931
- * bounded read and can never be voiced (§2.3). */
939
+ * These are PROPOSALS only. Every candidate still faces the byte-exact prefix
940
+ * compare and all three guards below, so a wrong proposal costs one bounded
941
+ * read and can never be voiced (exact-vs-approximate.md). */
932
942
  export function formsOpenedBy(
933
943
  ctx: MindContext,
934
944
  query: Uint8Array,
package/src/mind/types.ts CHANGED
@@ -65,12 +65,38 @@ export interface GraphSearchHost {
65
65
  starts: ReadonlySet<number>;
66
66
  };
67
67
  chooseNext?(node: number): number | undefined;
68
+ /** The boundary positions of `bytes` under the engine's ONE boundary rule
69
+ * (geometry.ts's `contentBoundaries`), or undefined when the host has no
70
+ * space to ask. The join's key is an entity plus a prefix of the tail, and
71
+ * the prefix that names a stored relation ENDS on one of these boundaries —
72
+ * measured, 5 of 5 accepted keys over four join-firing queries, where the
73
+ * byte-by-byte scan spent 153 probes for 14 boundaries. Boundaries are
74
+ * content-defined and STABLE under prefix extension, which is why a corpus
75
+ * key's end is a boundary of the query's own fold of the same bytes. */
76
+ contentCuts?(bytes: Uint8Array): readonly number[];
68
77
  /** The admission predicate — `traverse.ts`'s `leadsSomewhere`, its ONE
69
78
  * definition: does this node bear an edge or a halo? Optional, so a bare
70
79
  * host (a raw Store and nothing else) still works; when present, the search
71
80
  * uses it rather than re-probing the store, which keeps the predicate
72
81
  * single-defined AND memoised on the response-scoped struct cache. */
73
82
  leadsSomewhere?(id: number): boolean;
83
+ /** Report a SEARCH REFUSAL into the rationale — the channel AGENTS §6
84
+ * requires: a callback threaded through a call chain must FEED the
85
+ * rationale, the way `GraphSearch`'s `onDerivation` feeds `traceDerivation`,
86
+ * never a channel of its own. Optional, so a bare host stays silent rather
87
+ * than crashing. */
88
+ reportSearch?(
89
+ name: string,
90
+ parts: ReadonlyArray<Uint8Array>,
91
+ note: string,
92
+ ): void;
93
+ /** The CANONICAL resolver ({@link canonResolve}), optional like
94
+ * {@link leadsSomewhere}. The store's keys were written through the
95
+ * canonical fold, so a fact's `Gustaf Molander` and the deposited
96
+ * `gustaf molander` are the SAME node (measured inside a response: the
97
+ * canonical resolver maps the surface form to the deposited node while a raw
98
+ * resolve returns null). A bare host falls back to the plain probe. */
99
+ canonResolve?(bytes: Uint8Array): number | null;
74
100
  }
75
101
 
76
102
  // ═══════════════════════════════════════════════════════════════════════════
@@ -297,10 +323,10 @@ export type AItem =
297
323
  export interface MindContext extends GraphSearchHost {
298
324
  store: Store;
299
325
  /** The work accumulator for the inference call in flight, or null when
300
- * nothing is profiling — see src/meter.ts. WRITE-ONLY from the engine's
301
- * point of view: no inference decision may read a counter, or the
302
- * determinism contract (AGENTS §2.1) is gone. Every call site is
303
- * `ctx.meter?.x++`, so an unprofiled response allocates nothing. */
326
+ * nothing is profiling — see src/meter.ts. WRITE-ONLY from the engine's point
327
+ * of view: no inference decision may read a counter, or determinism is gone
328
+ * (determinism.md). Every call site is `ctx.meter?.x++`, so an unprofiled
329
+ * response allocates nothing. */
304
330
  meter: Meter | null;
305
331
  space: Space;
306
332
  alphabet: Alphabet;
package/src/store.ts CHANGED
@@ -541,16 +541,16 @@ export interface Store {
541
541
  // selected by identity hash so the choice is a property of each constituent
542
542
  // and never of where it sits in the fold.
543
543
  //
544
- // DURABLE DERIVED STATE, NOT A CACHE. §2.12 permits a cache to cost only
544
+ // DURABLE DERIVED STATE, NOT A CACHE. caches.md permits a cache to cost only
545
545
  // speed; this decides which terms enter a halo — a learned relation — so an
546
- // eviction would change the geometry rather than slow it down. It is
546
+ // eviction would change the geometry rather than slow it down. It is
547
547
  // therefore written like the canon index: computed once, kept, never
548
- // budgeted. Soundness rests on the set being INTRINSIC — minimality,
549
- // `len ≥ W` and non-domination are properties of the node's own subtree and
550
- // do not move as the corpus grows. The one corpus-dependent reading, the
551
- // hub exclusion, is deliberately NOT stored: it is applied by the caller at
552
- // pour time over the ≤ k candidates, which is the drift companyProfile
553
- // already documents as benign and one-directional.
548
+ // budgeted. Soundness rests on the set being INTRINSIC — minimality, `len ≥
549
+ // W` and non-domination are properties of the node's own subtree and do not
550
+ // move as the corpus grows. The one corpus-dependent reading, the hub
551
+ // exclusion, is deliberately NOT stored: it is applied by the caller at pour
552
+ // time over the ≤ k candidates, which is the drift companyProfile already
553
+ // documents as benign and one-directional.
554
554
  //
555
555
  // Backends that do not implement the pair leave both absent; companyProfile
556
556
  // then recomputes the sketch per pour and simply loses the amortisation.
@@ -1789,18 +1789,18 @@ export abstract class AbstractStore implements Store {
1789
1789
  * remainders must fit the budget. Scattered differences leave a wide
1790
1790
  * middle and are rejected.
1791
1791
  *
1792
- * Every read here is CAPPED (§2.8). It used to open with
1793
- * `bytesPrefix(k, Number.MAX_SAFE_INTEGER)` — the ALL sentinel, i.e. the
1794
- * full materialising `bytes()` read — on the deposit hot path, and only
1795
- * then compare lengths. So a candidate the length test was about to reject
1796
- * had already been reconstructed byte for byte. The LENGTHS decide first
1797
- * instead, from the `contentLen` memo the interning order has already built
1798
- * bottom-up, and the target's length is itself read under a cap: a target
1799
- * longer than `la + W` is rejected without touching one of its bytes.
1800
- * Same semantics — the old capped `b` read would have produced
1801
- * `a.length + W + 1` here and failed the very same test — strictly fewer
1802
- * byte reads. The `+ 1` on each byte cap keeps `_prefix`'s
1803
- * "complete reconstruction" test true, so the results still cache. */
1792
+ * Every read here is CAPPED (bounded-reads.md). It used to open with
1793
+ * `bytesPrefix(k, Number.MAX_SAFE_INTEGER)` — the ALL sentinel, i.e. the full
1794
+ * materialising `bytes()` read — on the deposit hot path, and only then
1795
+ * compare lengths. So a candidate the length test was about to reject had
1796
+ * already been reconstructed byte for byte. The LENGTHS decide first instead,
1797
+ * from the `contentLen` memo the interning order has already built bottom-up,
1798
+ * and the target's length is itself read under a cap: a target longer than
1799
+ * `la + W` is rejected without touching one of its bytes. Same semantics —
1800
+ * the old capped `b` read would have produced `a.length + W + 1` here and
1801
+ * failed the very same test — strictly fewer byte reads. The `+ 1` on each
1802
+ * byte cap keeps `_prefix`'s "complete reconstruction" test true, so the
1803
+ * results still cache. */
1804
1804
  private differsByOneWindow(
1805
1805
  kids: NodeId[],
1806
1806
  targetId: NodeId,
@@ -361,7 +361,7 @@ test("a halo accumulates poured signatures and gates on mass", async () => {
361
361
  });
362
362
 
363
363
  // A multi-turn conversation is deposited as ACCUMULATED-CONTEXT episodes — the
364
- // pattern HOW_IT_WORKS §19a prescribes and example/train.ts uses:
364
+ // pattern example/train.ts uses:
365
365
  // (t0) → t1
366
366
  // (t0 + t1) → t2
367
367
  // (t0 + t1 + t2) → t3
@@ -0,0 +1,109 @@
1
+ // 100-complete-grounding-trace.test.mjs — a declared-complete grounding SAYS
2
+ // SO in the rationale.
3
+ //
4
+ // THE GAP THIS CLOSES. `pipeline.ts` ends the derivation when the winning
5
+ // grounding carries `MechanismResult.complete` — the mechanism's own claim that
6
+ // the query IS a stored context, so its continuation is the whole read-out and
7
+ // a further pivot could only chain PAST the fact that produced the answer. The
8
+ // decision was correct and SILENT: no step, no note. A reader of the rationale
9
+ // therefore could not tell
10
+ //
11
+ // "the chain stopped because the query WAS the context" (complete)
12
+ //
13
+ // apart from
14
+ //
15
+ // "nothing followed" (no continuation)
16
+ //
17
+ // which are different claims about the same answer. AGENTS §6 makes that an
18
+ // instrumentation defect rather than a documentation gap: a bound that
19
+ // truncates must be reportable AT THE POINT it truncates, through the one
20
+ // surface that already exists.
21
+ //
22
+ // WHAT IS PINNED HERE.
23
+ // 1. the stop is REPORTED (step `completeGrounding`) when it happens;
24
+ // 2. the stop is REAL — the post-grounding extension did not run (no pivot or
25
+ // forward-absorb step), so the report cannot rot into a lie;
26
+ // 3. the report does NOT appear for an ordinary grounding, so it names a
27
+ // decision rather than decorating every response.
28
+ //
29
+ // The fixture is test/76's CARRIED frame: the continuation quotes its filler
30
+ // (`Run gcc <X>`), which is the shape the reference mechanism binds and
31
+ // declares complete.
32
+
33
+ import { test } from "node:test";
34
+ import assert from "node:assert/strict";
35
+ import { Mind } from "../dist/src/index.js";
36
+ import { SQliteStore } from "../dist/src/store-sqlite.js";
37
+
38
+ const CARRIED = [
39
+ ["How do I compile hello.c?", "Run gcc hello.c"],
40
+ ["How do I compile server.c?", "Run gcc server.c"],
41
+ ["How do I compile parser.c?", "Run gcc parser.c"],
42
+ ];
43
+
44
+ /** The frame fixture — the same one test/76 pins the binding on. */
45
+ async function frame() {
46
+ const m = new Mind({ seed: 7, store: new SQliteStore({ path: ":memory:" }) });
47
+ await m.ingest(CARRIED);
48
+ return m;
49
+ }
50
+
51
+ /** A grounding that is NOT declared complete: one plain learnt edge. */
52
+ async function plainChain() {
53
+ const m = new Mind({ seed: 7, store: new SQliteStore({ path: ":memory:" }) });
54
+ await m.ingest([["Eva director", "The director of Eva is Gustaf Molander."]]);
55
+ return m;
56
+ }
57
+
58
+ const moves = (steps) => steps.map((s) => s.mechanism.at(-1));
59
+
60
+ test("a binding that declares itself complete reports the stop", async () => {
61
+ const m = await frame();
62
+ const steps = [];
63
+ await m.respondText("How do I compile main.c?", (s) => steps.push(s));
64
+
65
+ const stop = steps.find((s) => s.mechanism.at(-1) === "completeGrounding");
66
+ assert.ok(stop, "the rationale must report the declared-complete stop");
67
+ assert.match(
68
+ stop.note,
69
+ /declared complete/,
70
+ "the note must carry the CLAIM, not a description of the answer",
71
+ );
72
+ });
73
+
74
+ test("the reported stop is real: the extension did not run", async () => {
75
+ const m = await frame();
76
+ const steps = [];
77
+ const answer = await m.respondText(
78
+ "How do I compile main.c?",
79
+ (s) => steps.push(s),
80
+ );
81
+
82
+ // The binding spliced the asker's referent (behaviour under test, asserted
83
+ // WITHOUT a trace below — a trace can change an answer, so the step and the
84
+ // bytes are pinned separately).
85
+ assert.ok(answer.length > 0, "the binding must answer");
86
+ const seen = moves(steps);
87
+ assert.equal(
88
+ seen.includes("pivotStep") || seen.includes("absorbForward"),
89
+ false,
90
+ "a complete grounding must not be extended",
91
+ );
92
+ });
93
+
94
+ test("the non-traced answer is the spliced one", async () => {
95
+ const m = await frame();
96
+ const answer = await m.respondText("How do I compile main.c?");
97
+ assert.equal(answer.replace(/\0+/g, "").trim(), "Run gcc main.c");
98
+ });
99
+
100
+ test("an ordinary grounding reports no complete-grounding stop", async () => {
101
+ const m = await plainChain();
102
+ const steps = [];
103
+ await m.respondText("Eva director", (s) => steps.push(s));
104
+ assert.equal(
105
+ moves(steps).includes("completeGrounding"),
106
+ false,
107
+ "the step names a decision, not every response",
108
+ );
109
+ });
@@ -0,0 +1,106 @@
1
+ // 101-alignment-gap-bound.test.mjs — the alignment's gap bound SAYS SO when it
2
+ // bites.
3
+ //
4
+ // THE GAP THIS CLOSES. `alignAround` matches a query against a stored context
5
+ // by sweeping for the next common run of ≥ W bytes, with each side's gap
6
+ // bounded by `chainReach(W)` = W². When the sweep ends with material still
7
+ // unmatched on BOTH sides, the alignment stopped because of that BOUND — and
8
+ // nothing reported it: `match.ts` emitted no trace at all. A reader of the
9
+ // rationale therefore saw a substituted answer (the frame filled with ANOTHER
10
+ // instance's filler) with no way to tell that the answer's own span had been
11
+ // refused by a cap.
12
+ //
13
+ // MEASURED BOUNDARY (W = 4, cap = 16). Against a learned
14
+ // `Book a table at <filler> tonight.` → `Your table at <filler> is booked.`
15
+ // frame, a NOVEL filler binds while it fits the cap and is replaced by a stored
16
+ // instance's filler once it does not:
17
+ //
18
+ // filler 7 B → answer quotes the asker's filler (bound)
19
+ // filler 12 B → answer quotes the asker's filler (bound)
20
+ // filler 18 B → answer carries ANOTHER filler (substituted)
21
+ // filler 30 B → answer carries ANOTHER filler (substituted)
22
+ //
23
+ // THIS FILE PINS THE REPORT, not the substitution: the step must exist and must
24
+ // carry the derived cap, so the bound is auditable at the point it truncates
25
+ // (AGENTS §6). The substitution itself is the scope defect a later lot fixes;
26
+ // when it is fixed, this file's behavioural control (a filler inside the cap is
27
+ // quoted back) stays true and the report stays reachable.
28
+
29
+ import { test } from "node:test";
30
+ import assert from "node:assert/strict";
31
+ import { Mind } from "../dist/src/index.js";
32
+ import { alignAround } from "../dist/src/mind/match.js";
33
+ import { SQliteStore } from "../dist/src/store-sqlite.js";
34
+
35
+ const WORDS =
36
+ ("alpha bravo charlie delta echo foxtrot golf hotel india juliet kilo lima " +
37
+ "mike november oscar papa quebec romeo sierra tango uniform victor whiskey " +
38
+ "xray yankee zulu amber bronze copper dahlia ember fjord gossamer harbour " +
39
+ "indigo jasmine kestrel lantern marigold nectar opal pewter quartz ripple " +
40
+ "saffron thistle umber violet willow xenon yarrow").split(" ");
41
+
42
+ const short = (i) => `${WORDS[i % 50]}${i}`;
43
+ const long = (i) =>
44
+ `${WORDS[i % 50]} ${WORDS[(i * 3) % 50]} ${WORDS[(i * 7) % 50]} street ${i}`;
45
+
46
+ /** The frame whose continuation quotes its filler, with mixed filler widths. */
47
+ async function frame(n = 60, opts = {}) {
48
+ const store = new SQliteStore({ path: ":memory:", D: 1024 });
49
+ const mind = new Mind({ seed: 7, store, ...opts });
50
+ const pairs = [];
51
+ for (let i = 0; i < n; i++) {
52
+ const f = i % 3 === 0 ? long(i) : short(i);
53
+ pairs.push([
54
+ `Book a table at ${f} tonight.`,
55
+ `Your table at ${f} is booked.`,
56
+ ]);
57
+ }
58
+ await mind.ingest(pairs);
59
+ return { store, mind };
60
+ }
61
+
62
+ /** A novel filler of exactly `len` bytes, never ingested. */
63
+ const novel = (len) => "zephyr quartz lantern ".repeat(3).slice(0, len).trim();
64
+
65
+ // THE LAW'S ASSERTION, straight on the aligner. There is no bound to report
66
+ // any more: the sweep's work is proportional to the bytes a run spans, so a
67
+ // divergence far past the old arity bound (chainReach(W) = 16) is simply
68
+ // bridged, on both sides, by finding the next common run.
69
+ test("a divergence far past the old arity bound is bridged, not truncated", () => {
70
+ const enc = new TextEncoder();
71
+ const head = "the common head of the frame ";
72
+ const tail = " and the common tail of the frame";
73
+ const q = enc.encode(head + "a".repeat(60) + tail);
74
+ const c = enc.encode(head + "b".repeat(60) + tail);
75
+ const at = head.length - 1; // the seed: the shared head's own boundary
76
+ const { matched, gaps } = alignAround(
77
+ { space: { maxGroup: 4 } },
78
+ q,
79
+ c,
80
+ at,
81
+ at,
82
+ );
83
+ assert.equal(
84
+ matched.length,
85
+ 2,
86
+ "both common runs must be found: " + JSON.stringify(matched),
87
+ );
88
+ assert.equal(gaps.length, 1, "with exactly one substitution between them");
89
+ const g = gaps[0];
90
+ assert.ok(
91
+ g.qe - g.qs >= 60 && g.ce - g.cs >= 60,
92
+ "the substitution's extent is the pair's own: " + JSON.stringify(g),
93
+ );
94
+ });
95
+
96
+ test("control: a filler inside the cap is quoted back, and the fixture is live", async () => {
97
+ const filler = novel(12);
98
+ const { mind } = await frame();
99
+ const answer = (await mind.respondText(
100
+ `Book a table at ${filler} tonight.`,
101
+ )).replace(/\0+/g, "").trim();
102
+ assert.ok(
103
+ answer.includes(filler),
104
+ "inside the cap the frame binds the asker's own filler",
105
+ );
106
+ });