@hviana/sema 0.8.0 → 0.8.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (119) hide show
  1. package/AGENTS.md +22 -1
  2. package/DATASETS.md +1 -1
  3. package/dist/example/train_base/config.js +2 -2
  4. package/dist/example/train_base/corpora/massive.js +1 -1
  5. package/dist/example/train_base/readers.js +1 -1
  6. package/dist/src/config.d.ts +17 -0
  7. package/dist/src/config.js +18 -0
  8. package/dist/src/geometry.d.ts +10 -10
  9. package/dist/src/geometry.js +25 -24
  10. package/dist/src/meter.d.ts +29 -12
  11. package/dist/src/meter.js +58 -14
  12. package/dist/src/mind/attention.js +12 -12
  13. package/dist/src/mind/bridge.d.ts +8 -8
  14. package/dist/src/mind/bridge.js +33 -32
  15. package/dist/src/mind/corpus.d.ts +40 -0
  16. package/dist/src/mind/corpus.js +149 -0
  17. package/dist/src/mind/graph-search.d.ts +7 -8
  18. package/dist/src/mind/graph-search.js +244 -32
  19. package/dist/src/mind/index.d.ts +3 -1
  20. package/dist/src/mind/index.js +1 -0
  21. package/dist/src/mind/junction.d.ts +1 -1
  22. package/dist/src/mind/junction.js +8 -8
  23. package/dist/src/mind/learning.js +36 -35
  24. package/dist/src/mind/match.d.ts +8 -3
  25. package/dist/src/mind/match.js +156 -71
  26. package/dist/src/mind/mechanisms/cast.js +18 -2
  27. package/dist/src/mind/mechanisms/cover.js +19 -12
  28. package/dist/src/mind/mechanisms/prefix-completion.js +24 -24
  29. package/dist/src/mind/mechanisms/recall.js +38 -40
  30. package/dist/src/mind/mechanisms/reference.js +16 -16
  31. package/dist/src/mind/mind.d.ts +61 -7
  32. package/dist/src/mind/mind.js +72 -2
  33. package/dist/src/mind/pipeline-mechanism.d.ts +10 -8
  34. package/dist/src/mind/pipeline-mechanism.js +25 -21
  35. package/dist/src/mind/pipeline.d.ts +9 -9
  36. package/dist/src/mind/pipeline.js +49 -29
  37. package/dist/src/mind/primitives.d.ts +5 -5
  38. package/dist/src/mind/primitives.js +5 -5
  39. package/dist/src/mind/reasoning.d.ts +5 -1
  40. package/dist/src/mind/reasoning.js +54 -1
  41. package/dist/src/mind/recognition.d.ts +14 -13
  42. package/dist/src/mind/recognition.js +23 -23
  43. package/dist/src/mind/resonance.js +21 -21
  44. package/dist/src/mind/traverse.d.ts +54 -52
  45. package/dist/src/mind/traverse.js +83 -73
  46. package/dist/src/mind/types.d.ts +26 -4
  47. package/dist/src/store.d.ts +12 -12
  48. package/dist/src/store.js +12 -12
  49. package/docs/INDEX.md +2 -2
  50. package/docs/architecture/exact-vs-approximate.md +2 -1
  51. package/docs/architecture/fold-contract.md +1 -1
  52. package/docs/failures/tempting-but-wrong.md +33 -5
  53. package/docs/harness/gates.md +7 -7
  54. package/example/train_base/config.ts +2 -2
  55. package/example/train_base/corpora/massive.ts +1 -1
  56. package/example/train_base/readers.ts +1 -1
  57. package/jsr.json +1 -1
  58. package/package.json +1 -1
  59. package/src/config.ts +35 -0
  60. package/src/geometry.ts +25 -24
  61. package/src/meter.ts +61 -14
  62. package/src/mind/attention.ts +12 -12
  63. package/src/mind/bridge.ts +33 -32
  64. package/src/mind/corpus.ts +202 -0
  65. package/src/mind/graph-search.ts +261 -31
  66. package/src/mind/index.ts +8 -1
  67. package/src/mind/junction.ts +8 -8
  68. package/src/mind/learning.ts +36 -35
  69. package/src/mind/match.ts +163 -73
  70. package/src/mind/mechanisms/cast.ts +17 -1
  71. package/src/mind/mechanisms/cover.ts +18 -12
  72. package/src/mind/mechanisms/prefix-completion.ts +24 -24
  73. package/src/mind/mechanisms/recall.ts +38 -40
  74. package/src/mind/mechanisms/reference.ts +16 -16
  75. package/src/mind/mind.ts +129 -7
  76. package/src/mind/pipeline-mechanism.ts +25 -21
  77. package/src/mind/pipeline.ts +63 -38
  78. package/src/mind/primitives.ts +5 -5
  79. package/src/mind/reasoning.ts +55 -0
  80. package/src/mind/recognition.ts +23 -23
  81. package/src/mind/resonance.ts +21 -21
  82. package/src/mind/traverse.ts +83 -73
  83. package/src/mind/types.ts +30 -4
  84. package/src/store.ts +20 -20
  85. package/test/08-storage.test.mjs +1 -1
  86. package/test/100-complete-grounding-trace.test.mjs +109 -0
  87. package/test/101-alignment-gap-bound.test.mjs +106 -0
  88. package/test/102-production-composes-at-scale.test.mjs +110 -0
  89. package/test/103-alignment-gap-budget.test.mjs +89 -0
  90. package/test/104-composition-is-reported.test.mjs +90 -0
  91. package/test/105-derive-through-reports-its-refusal.test.mjs +113 -0
  92. package/test/106-the-join-fires.test.mjs +94 -0
  93. package/test/107-the-join-is-counted.test.mjs +81 -0
  94. package/test/108-the-join-chains.test.mjs +78 -0
  95. package/test/109-the-pivot-is-counted.test.mjs +60 -0
  96. package/test/110-the-reasoner-stops-when-the-question-is-answered.test.mjs +91 -0
  97. package/test/111-the-cover-assembly-is-counted.test.mjs +74 -0
  98. package/test/112-the-exploration-does-not-grow-with-the-hub.test.mjs +89 -0
  99. package/test/113-the-rationale-payload-is-bounded.test.mjs +84 -0
  100. package/test/114-alignment-budget-is-per-sweep.test.mjs +93 -0
  101. package/test/116-the-extension-is-gated-by-the-pipelines-own-remainder.test.mjs +100 -0
  102. package/test/117-corpus-search.test.mjs +171 -0
  103. package/test/14-scaling.test.mjs +10 -7
  104. package/test/35-prefix-edge.test.mjs +1 -1
  105. package/test/40-choosenext-scale-guard.test.mjs +16 -17
  106. package/test/56-bridge-identity-admission.test.mjs +6 -6
  107. package/test/70-prefix-completion.test.mjs +4 -3
  108. package/test/72-prefix-candidate-supply.test.mjs +3 -3
  109. package/test/73-scaffolding-only-bridge-abstains.test.mjs +6 -6
  110. package/test/75-multiturn-context-optimisation.test.mjs +5 -5
  111. package/test/76-reference-binding.test.mjs +6 -1
  112. package/test/84-composed-answer-honesty.test.mjs +5 -6
  113. package/test/88-dependency-footprint.test.mjs +1 -1
  114. package/test/89-completion-recursion.test.mjs +47 -19
  115. package/test/90-connector-read-cap.test.mjs +10 -8
  116. package/test/93-regime-prediction.test.mjs +10 -10
  117. package/test/94-cross-region-budget.test.mjs +2 -2
  118. package/test/95-wide-resonance-removed.test.mjs +8 -7
  119. package/test/96-bytes-walk-termination.test.mjs +3 -3
@@ -9,11 +9,11 @@ import { cosine } from "../vec.js";
9
9
  import { gistOf, read } from "./primitives.js";
10
10
  import { canonicalWindows, leafIdPrefix, leafIdRun } from "./canonical.js";
11
11
  //
12
- // Budgeted on the same terms as the reach memo below (AGENTS §2.12): these
13
- // three maps are cleared on every write, but a long read-only session over a
14
- // large store converges on one entry per node per map with nothing to bound
15
- // it. Past the cap all three are dropped together and re-derived, costing
16
- // cold structural probes and never a wrong answer.
12
+ // Budgeted on the same terms as the reach memo below (caches.md): these three
13
+ // maps are cleared on every write, but a long read-only session over a large
14
+ // store converges on one entry per node per map with nothing to bound it. Past
15
+ // the cap all three are dropped together and re-derived, costing cold
16
+ // structural probes and never a wrong answer.
17
17
  const STRUCT_MEMO_MAX = 100_000;
18
18
  const structCaches = new WeakMap();
19
19
  // ── The shared ancestor-reach memo ──────────────────────────────────────
@@ -31,20 +31,21 @@ const structCaches = new WeakMap();
31
31
  // battery repeatedly reaches the same corpus scaffolding even when its
32
32
  // surface questions differ.
33
33
  //
34
- // Budgeted, not unbounded (AGENTS §2.12): past the cap the whole map is
35
- // dropped and re-derived, costing a cold climb and never a wrong answer.
34
+ // Budgeted, not unbounded (caches.md): past the cap the whole map is dropped
35
+ // and
36
+ // re-derived, costing a cold climb and never a wrong answer.
36
37
  const REACH_MEMO_MAX = 100_000;
37
38
  const reachCaches = new WeakMap();
38
39
  /** The reach memo this ask should use — see the note above.
39
40
  *
40
- * A TRACED response always gets a fresh, empty one. `AncestorReach`'s
41
- * `visited`/`maxDepth`/`saturation` fields are populated only when a trace
42
- * is attached, so an entry deposited by an untraced earlier turn would
43
- * silently black out the reach detail of a later traced one; and the trace's
44
- * reach payload is serialised by ITERATING this map, which must therefore
45
- * hold what THIS climb consulted, not the whole conversation's history.
46
- * Consistent with AGENTS §2.11: a traced response is a different machine —
47
- * never benchmark with a trace attached. */
41
+ * A TRACED response always gets a fresh, empty one. `AncestorReach`'s
42
+ * `visited`/`maxDepth`/`saturation` fields are populated only when a trace is
43
+ * attached, so an entry deposited by an untraced earlier turn would silently
44
+ * black out the reach detail of a later traced one; and the trace's reach
45
+ * payload is serialised by ITERATING this map, which must therefore hold what
46
+ * THIS climb consulted, not the whole conversation's history. Consistent with
47
+ * memoization.md: a traced response is a different machine — never benchmark
48
+ * with a trace attached. */
48
49
  export function sharedReachMemo(ctx) {
49
50
  if (ctx.trace !== null || ctx.climbMemo === null)
50
51
  return new Map();
@@ -426,13 +427,14 @@ export function bearsEdge(ctx, id) {
426
427
  return cachedHasNext(ctx, id, getStructCache(ctx));
427
428
  }
428
429
  /** Whether a node LEADS SOMEWHERE — it bears a continuation edge or a halo.
429
- * The admission predicate recognition filters sites with (HOW_IT_WORKS
430
- * §15.3): a form that leads nowhere contributes nothing to any derivation.
431
- * Runs once per candidate span on the recognition hot path — `hasNext` is
432
- * cached per response (the same flat-branch ids are probed across prefix
433
- * variants by canonicalChunkId). `hasHalo` is not cached: it's a single
434
- * indexed point probe per candidate, and the candidates that reach this
435
- * check have already been filtered by hasNext above in edgeAncestors. */
430
+ * The admission predicate recognition filters sites with (cover.md): a form
431
+ * that
432
+ * leads nowhere contributes nothing to any derivation. Runs once per candidate
433
+ * span on the recognition hot path — `hasNext` is cached per response (the same
434
+ * flat-branch ids are probed across prefix variants by canonicalChunkId).
435
+ * `hasHalo` is not cached: it's a single indexed point probe per candidate, and
436
+ * the candidates that reach this check have already been filtered by hasNext
437
+ * above in edgeAncestors. */
436
438
  export function leadsSomewhere(ctx, id) {
437
439
  const memo = getStructCache(ctx);
438
440
  if (cachedHasNext(ctx, id, memo))
@@ -476,10 +478,10 @@ function boundFor(contextCount) {
476
478
  return Math.ceil(Math.sqrt(Math.max(2, contextCount)));
477
479
  }
478
480
  /** Cap a candidate list at the hub bound √N (insertion order) — the ONE
479
- * fan-out convention every walk and disambiguation uses (see HOW_IT_WORKS
480
- * §8.6). A node connected to more than √N others is a hub whose individual
481
- * connections carry ~no discriminative information; materialising or scoring
482
- * them all would make single decisions scale with the corpus. */
481
+ * fan-out convention every walk and disambiguation uses (see bounded-reads.md).
482
+ * A node connected to more than √N others is a hub whose individual connections
483
+ * carry ~no discriminative information; materialising or scoring them all would
484
+ * make single decisions scale with the corpus. */
483
485
  export function hubCap(ctx, ids) {
484
486
  const bound = hubBound(ctx);
485
487
  return ids.length > bound ? ids.slice(0, bound) : ids;
@@ -512,16 +514,16 @@ export function contains(ctx, ancestor, descendant) {
512
514
  * the EXACT half's veto on calling them synonyms.
513
515
  *
514
516
  * Halos measure company, and the strongest company any two forms can keep is
515
- * standing next to each other: a question and its answer co-occur in every
516
- * episode that taught the pair, so their halos SHOULD be similar, and on a
517
- * conversational store they are (measured on the CONV fixture: consecutive
518
- * turns at 0.809 against a 0.516 concept threshold). A gate reading halo
519
- * cosine alone therefore reads adjacency as synonymy and revoices an answer
520
- * in the words of the question it answers — "it hangs in madrid" spliced back
521
- * into "where is it kept now". The distributional layer cannot tell the two
522
- * relations apart, because to it they are the same observation; the exact
523
- * half can, for free, because it stored the edge. §4.1's division of labour
524
- * exactly: approximate proposes, exact decides.
517
+ * standing next to each other: a question and its answer co-occur in every
518
+ * episode that taught the pair, so their halos SHOULD be similar, and on a
519
+ * conversational store they are (measured on the CONV fixture: consecutive
520
+ * turns at 0.809 against a 0.516 concept threshold). A gate reading halo cosine
521
+ * alone therefore reads adjacency as synonymy and revoices an answer in the
522
+ * words of the question it answers — "it hangs in madrid" spliced back into
523
+ * "where is it kept now". The distributional layer cannot tell the two
524
+ * relations apart, because to it they are the same observation; the exact half
525
+ * can, for free, because it stored the edge. halo-sketch.md's division of
526
+ * labour exactly: approximate proposes, exact decides.
525
527
  *
526
528
  * Read LIMITed in both directions at the hub bound — a common continuation's
527
529
  * fan-in is corpus-sized, and no single decision may scale with it. */
@@ -645,25 +647,32 @@ export function chooseNext(ctx, id, guide) {
645
647
  // NO consensusFloor gate here (tried and reverted — see
646
648
  // test/40-choosenext-scale-guard.test.mjs): that floor is calibrated for
647
649
  // POOLED, IDF-weighted CLIMB VOTES (recallByResonance, commitVotes), where
648
- // each corroborating region contributes at most ln N and the floor grows
649
- // with N exactly as that per-region ceiling does (HOW_IT_WORKS.md §8.6).
650
- // `bestSupport` here is a different kind of quantity — a raw prevCount of
651
- // how many training contexts predicted ONE destination, bounded by how
652
- // often that specific fact was retold, never by corpus size N. Gating an
653
- // N-invariant count against an N-growing threshold guarantees failure
654
- // once N is large enough, discarding genuinely, structurally dominant
655
- // edges (observed: a fact corroborated 2-to-1-1-1 refused at N≈325K,
656
- // falling back to a noisy concept-hop). The loop above already IS the
657
- // "genuinely competing" test: a tie leaves first-inserted as the pick
658
- // (test/30's own pinned behaviour); a strict winner is real evidence
659
- // regardless of corpus scale. Matches HOW_IT_WORKS.md §25's own
660
- // chooseNext pseudocode, which has no such floor.
650
+ // each corroborating region contributes at most ln N and the floor grows with
651
+ // N exactly as that per-region ceiling does (thresholds.md). `bestSupport`
652
+ // here is a different kind of quantity — a raw prevCount of how many training
653
+ // contexts predicted ONE destination, bounded by how often that specific fact
654
+ // was retold, never by corpus size N. Gating an N-invariant count against an
655
+ // N-growing threshold guarantees failure once N is large enough, discarding
656
+ // genuinely, structurally dominant edges (observed: a fact corroborated
657
+ // 2-to-1-1-1 refused at N≈325K, falling back to a noisy concept-hop). The
658
+ // loop above already IS the "genuinely competing" test: a tie leaves
659
+ // first-inserted as the pick (test/30's own pinned behaviour); a strict
660
+ // winner is real evidence regardless of corpus scale. Matches `chooseNext`'s
661
+ // own pseudocode, which has no such floor.
661
662
  // Trace is built lazily — the filter + map below only execute when a
662
663
  // trace listener is attached, so the common (no-trace) path pays only
663
664
  // for the prevCount calls in the loop above, never for extra rItemShort
664
665
  // byte-reads.
665
666
  if (ctx.trace) {
666
- const others = capped.filter((c) => c !== best);
667
+ // A BOUNDED SAMPLE, AND THE COUNT. The step used to carry EVERY candidate
668
+ // it weighed — measured on the trained store, 1559 out-items in one step
669
+ // (hubBound's own size) and 1082 in another (the hub's degree). The
670
+ // rationale's job is to explain the CHOICE, and the count is what says how
671
+ // wide the field was; the declared candidate budget (`recallQueryK`) is what
672
+ // bounds the sample, so no number is invented here.
673
+ const others = capped
674
+ .filter((c) => c !== best)
675
+ .slice(0, ctx.cfg.rationaleSampleK);
667
676
  ctx.trace.step("disambiguate", [rItemShort(ctx, best, "halo-evidence", bestSupport)], others.map((c) => rItemShort(ctx, c, "candidate", ctx.store.prevCount(c))), `${capped.length} continuations — distributional evidence selects ` +
668
677
  `the most corroborated (distinct contexts ${bestSupport}, ` +
669
678
  `poured mass ${bestMass})`);
@@ -709,12 +718,13 @@ function rItemShort(ctx, id, role, score) {
709
718
  * W-window it spells is contained by more places than the hub bound allows,
710
719
  * i.e. the whole query is corpus-global scaffolding.
711
720
  *
712
- * WHAT IT IS FOR. Several mechanisms ground a query through the literal
713
- * spans it did NOT explain, and those spans are the whole of their evidence.
714
- * When every one of them is a hub, the query says nothing the corpus can be
715
- * held to, and grounding it means picking one of thousands of continuations
716
- * it gives no evidence for — a fabrication whatever the answer happens to be.
717
- * Answering with silence there is the honest degradation contract (§2.13).
721
+ * WHAT IT IS FOR. Several mechanisms ground a query through the literal spans
722
+ * it
723
+ * did NOT explain, and those spans are the whole of their evidence. When every
724
+ * one of them is a hub, the query says nothing the corpus can be held to, and
725
+ * grounding it means picking one of thousands of continuations it gives no
726
+ * evidence for — a fabrication whatever the answer happens to be. Answering
727
+ * with silence there is the honest degradation contract (INVARIANTS.md).
718
728
  *
719
729
  * MEASURED SEPARATION (trained store, hubBound 571) — this is categorical,
720
730
  * not marginal, and it is why the predicate lives here rather than being
@@ -731,11 +741,11 @@ function rItemShort(ctx, id, role, score) {
731
741
  * evidence and sit on the SAME side as the correct ones, so this predicate
732
742
  * is not what makes them silent and cannot be credited for them.
733
743
  *
734
- * NO NEW THRESHOLD (§2.2): `hubBound` is the √N reading of "hub" used
735
- * everywhere, and the containment read is clamped to it exactly as every
736
- * other fan-out read is (§2.8). A query with no stored window at all is NOT
737
- * scaffolding-only — it has no evidence either way, and its callers already
738
- * refuse it on their own terms. */
744
+ * NO NEW THRESHOLD (thresholds.md): `hubBound` is the √N reading of "hub" used
745
+ * everywhere, and the containment read is clamped to it exactly as every other
746
+ * fan-out read is (bounded-reads.md). A query with no stored window at all is
747
+ * NOT scaffolding-only — it has no evidence either way, and its callers already
748
+ * refuse it on their own terms. */
739
749
  export function allWindowsAreScaffolding(ctx, query) {
740
750
  const W = ctx.space.maxGroup;
741
751
  const bound = hubBound(ctx);
@@ -786,19 +796,19 @@ export function allWindowsAreScaffolding(ctx, query) {
786
796
  * by climbing containment then parents. Nothing is added to the write side;
787
797
  * this reads an index training already built.
788
798
  *
789
- * BOUNDED (§2.8), AND WITH NO NEW THRESHOLD. The window whose containment is
790
- * SMALLEST carries the most evidence, and one saturated at `hubBound` carries
791
- * none — that is the same √N reading of "hub" the rest of the mind uses, not
792
- * a tuned knob. The upward walk spends a budget of `hubBound` nodes and
793
- * fans out by W, so a hub query enumerates nothing and the caller stays
794
- * silent rather than guessing (§2.13). Measured on the trained store: the
795
- * photosynthesis form at a one-byte truncation picks a window with 52
796
- * containers, visits 446 nodes, and yields exactly ONE candidate that
797
- * survives the caller's byte compare — the form itself.
799
+ * BOUNDED (bounded-reads.md), AND WITH NO NEW THRESHOLD. The window whose
800
+ * containment is SMALLEST carries the most evidence, and one saturated at
801
+ * `hubBound` carries none — that is the same √N reading of "hub" the rest of
802
+ * the mind uses, not a tuned knob. The upward walk spends a budget of
803
+ * `hubBound` nodes and fans out by W, so a hub query enumerates nothing and the
804
+ * caller stays silent rather than guessing (INVARIANTS.md). Measured on the
805
+ * trained store: the photosynthesis form at a one-byte truncation picks a
806
+ * window with 52 containers, visits 446 nodes, and yields exactly ONE candidate
807
+ * that survives the caller's byte compare — the form itself.
798
808
  *
799
- * These are PROPOSALS only. Every candidate still faces the byte-exact
800
- * prefix compare and all three guards below, so a wrong proposal costs one
801
- * bounded read and can never be voiced (§2.3). */
809
+ * These are PROPOSALS only. Every candidate still faces the byte-exact prefix
810
+ * compare and all three guards below, so a wrong proposal costs one bounded
811
+ * read and can never be voiced (exact-vs-approximate.md). */
802
812
  export function formsOpenedBy(ctx, query) {
803
813
  const store = ctx.store;
804
814
  const W = ctx.space.maxGroup;
@@ -35,12 +35,34 @@ export interface GraphSearchHost {
35
35
  starts: ReadonlySet<number>;
36
36
  };
37
37
  chooseNext?(node: number): number | undefined;
38
+ /** The boundary positions of `bytes` under the engine's ONE boundary rule
39
+ * (geometry.ts's `contentBoundaries`), or undefined when the host has no
40
+ * space to ask. The join's key is an entity plus a prefix of the tail, and
41
+ * the prefix that names a stored relation ENDS on one of these boundaries —
42
+ * measured, 5 of 5 accepted keys over four join-firing queries, where the
43
+ * byte-by-byte scan spent 153 probes for 14 boundaries. Boundaries are
44
+ * content-defined and STABLE under prefix extension, which is why a corpus
45
+ * key's end is a boundary of the query's own fold of the same bytes. */
46
+ contentCuts?(bytes: Uint8Array): readonly number[];
38
47
  /** The admission predicate — `traverse.ts`'s `leadsSomewhere`, its ONE
39
48
  * definition: does this node bear an edge or a halo? Optional, so a bare
40
49
  * host (a raw Store and nothing else) still works; when present, the search
41
50
  * uses it rather than re-probing the store, which keeps the predicate
42
51
  * single-defined AND memoised on the response-scoped struct cache. */
43
52
  leadsSomewhere?(id: number): boolean;
53
+ /** Report a SEARCH REFUSAL into the rationale — the channel AGENTS §6
54
+ * requires: a callback threaded through a call chain must FEED the
55
+ * rationale, the way `GraphSearch`'s `onDerivation` feeds `traceDerivation`,
56
+ * never a channel of its own. Optional, so a bare host stays silent rather
57
+ * than crashing. */
58
+ reportSearch?(name: string, parts: ReadonlyArray<Uint8Array>, note: string): void;
59
+ /** The CANONICAL resolver ({@link canonResolve}), optional like
60
+ * {@link leadsSomewhere}. The store's keys were written through the
61
+ * canonical fold, so a fact's `Gustaf Molander` and the deposited
62
+ * `gustaf molander` are the SAME node (measured inside a response: the
63
+ * canonical resolver maps the surface form to the deposited node while a raw
64
+ * resolve returns null). A bare host falls back to the plain probe. */
65
+ canonResolve?(bytes: Uint8Array): number | null;
44
66
  }
45
67
  export interface Recognition {
46
68
  /** Forms that can lead somewhere — they have an edge or a halo. */
@@ -250,10 +272,10 @@ export type AItem = {
250
272
  export interface MindContext extends GraphSearchHost {
251
273
  store: Store;
252
274
  /** The work accumulator for the inference call in flight, or null when
253
- * nothing is profiling — see src/meter.ts. WRITE-ONLY from the engine's
254
- * point of view: no inference decision may read a counter, or the
255
- * determinism contract (AGENTS §2.1) is gone. Every call site is
256
- * `ctx.meter?.x++`, so an unprofiled response allocates nothing. */
275
+ * nothing is profiling — see src/meter.ts. WRITE-ONLY from the engine's point
276
+ * of view: no inference decision may read a counter, or determinism is gone
277
+ * (determinism.md). Every call site is `ctx.meter?.x++`, so an unprofiled
278
+ * response allocates nothing. */
257
279
  meter: Meter | null;
258
280
  space: Space;
259
281
  alphabet: Alphabet;
@@ -701,18 +701,18 @@ export declare abstract class AbstractStore implements Store {
701
701
  * remainders must fit the budget. Scattered differences leave a wide
702
702
  * middle and are rejected.
703
703
  *
704
- * Every read here is CAPPED (§2.8). It used to open with
705
- * `bytesPrefix(k, Number.MAX_SAFE_INTEGER)` — the ALL sentinel, i.e. the
706
- * full materialising `bytes()` read — on the deposit hot path, and only
707
- * then compare lengths. So a candidate the length test was about to reject
708
- * had already been reconstructed byte for byte. The LENGTHS decide first
709
- * instead, from the `contentLen` memo the interning order has already built
710
- * bottom-up, and the target's length is itself read under a cap: a target
711
- * longer than `la + W` is rejected without touching one of its bytes.
712
- * Same semantics — the old capped `b` read would have produced
713
- * `a.length + W + 1` here and failed the very same test — strictly fewer
714
- * byte reads. The `+ 1` on each byte cap keeps `_prefix`'s
715
- * "complete reconstruction" test true, so the results still cache. */
704
+ * Every read here is CAPPED (bounded-reads.md). It used to open with
705
+ * `bytesPrefix(k, Number.MAX_SAFE_INTEGER)` — the ALL sentinel, i.e. the full
706
+ * materialising `bytes()` read — on the deposit hot path, and only then
707
+ * compare lengths. So a candidate the length test was about to reject had
708
+ * already been reconstructed byte for byte. The LENGTHS decide first instead,
709
+ * from the `contentLen` memo the interning order has already built bottom-up,
710
+ * and the target's length is itself read under a cap: a target longer than
711
+ * `la + W` is rejected without touching one of its bytes. Same semantics —
712
+ * the old capped `b` read would have produced `a.length + W + 1` here and
713
+ * failed the very same test — strictly fewer byte reads. The `+ 1` on each
714
+ * byte cap keeps `_prefix`'s "complete reconstruction" test true, so the
715
+ * results still cache. */
716
716
  private differsByOneWindow;
717
717
  putLeaf(bytes: Uint8Array, gist: Vec): Promise<NodeId>;
718
718
  putBranch(kids: NodeId[], gist: Vec): Promise<NodeId>;
package/dist/src/store.js CHANGED
@@ -1235,18 +1235,18 @@ export class AbstractStore {
1235
1235
  * remainders must fit the budget. Scattered differences leave a wide
1236
1236
  * middle and are rejected.
1237
1237
  *
1238
- * Every read here is CAPPED (§2.8). It used to open with
1239
- * `bytesPrefix(k, Number.MAX_SAFE_INTEGER)` — the ALL sentinel, i.e. the
1240
- * full materialising `bytes()` read — on the deposit hot path, and only
1241
- * then compare lengths. So a candidate the length test was about to reject
1242
- * had already been reconstructed byte for byte. The LENGTHS decide first
1243
- * instead, from the `contentLen` memo the interning order has already built
1244
- * bottom-up, and the target's length is itself read under a cap: a target
1245
- * longer than `la + W` is rejected without touching one of its bytes.
1246
- * Same semantics — the old capped `b` read would have produced
1247
- * `a.length + W + 1` here and failed the very same test — strictly fewer
1248
- * byte reads. The `+ 1` on each byte cap keeps `_prefix`'s
1249
- * "complete reconstruction" test true, so the results still cache. */
1238
+ * Every read here is CAPPED (bounded-reads.md). It used to open with
1239
+ * `bytesPrefix(k, Number.MAX_SAFE_INTEGER)` — the ALL sentinel, i.e. the full
1240
+ * materialising `bytes()` read — on the deposit hot path, and only then
1241
+ * compare lengths. So a candidate the length test was about to reject had
1242
+ * already been reconstructed byte for byte. The LENGTHS decide first instead,
1243
+ * from the `contentLen` memo the interning order has already built bottom-up,
1244
+ * and the target's length is itself read under a cap: a target longer than
1245
+ * `la + W` is rejected without touching one of its bytes. Same semantics —
1246
+ * the old capped `b` read would have produced `a.length + W + 1` here and
1247
+ * failed the very same test — strictly fewer byte reads. The `+ 1` on each
1248
+ * byte cap keeps `_prefix`'s "complete reconstruction" test true, so the
1249
+ * results still cache. */
1250
1250
  differsByOneWindow(kids, targetId, W) {
1251
1251
  const lens = kids.map((k) => this.contentLen(k));
1252
1252
  let la = 0;
package/docs/INDEX.md CHANGED
@@ -1,8 +1,8 @@
1
1
  # Sema Documentation Index
2
2
 
3
3
  Sema is a single system stated three ways: the law lives in `docs/architecture/`
4
- (what holds), the prescription in `AGENTS.md` bootloader (what to do and where),
5
- and the proof in `test/` (pins that fail when the law is broken).
4
+ (what holds), the prescription in `AGENTS.md` (what to do and where), and the
5
+ proof in `test/` (pins that fail when the law is broken).
6
6
 
7
7
  ## Routing — what to read for each task
8
8
 
@@ -44,4 +44,5 @@ Two rules in `attention.ts` encode "exact decides" and must not be flattened:
44
44
 
45
45
  Add a tier to the shared family in `mind/match.ts` with a derived gate
46
46
  (`geometry.ts`), never a private `score >= k` check. A new mechanism is a
47
- `(matcher, direction, gate)` configuration over that family (§2.5).
47
+ `(matcher, direction, gate)` configuration over that family
48
+ (`docs/architecture/match-project.md`).
@@ -84,4 +84,4 @@ any change here.
84
84
 
85
85
  See:
86
86
  `src/geometry.ts:contentLevels`/`contentBoundaries`/`contentFoldIncremental`/`stablePrefixFold`;
87
- `AGENTS.md` bootloader invariants.
87
+ `AGENTS.md` §2 invariants.
@@ -1,6 +1,6 @@
1
- # Tempting but Wrong — 12 Traps
1
+ # Tempting but Wrong — 13 Traps
2
2
 
3
- Twelve shortcuts that look plausible and break an invariant. Each states what
3
+ Thirteen shortcuts that look plausible and break an invariant. Each states what
4
4
  not to do, why it fails, and what to do instead.
5
5
 
6
6
  ### 1. `score >= threshold` decides identity
@@ -76,7 +76,7 @@ not to do, why it fails, and what to do instead.
76
76
  - **WHY:** `frameSlots` reports (contracted gaps tagged
77
77
  substitution/insertion/deletion); `carriesFillers` judges;
78
78
  `Precomputed.frames` inventories — elects nothing (`AGENTS §2` Cross-cutting
79
- contracts — `docs/architecture/factored-machinery.md` § Frame reading).
79
+ contracts — `docs/architecture/match-project.md` § Frame reading).
80
80
  - **CORRECT:** Report everything in the shared layer; apply
81
81
  `substituteAll(contA, fillersA→fillersB)==contB` and
82
82
  frame-dominance/`W`-reach/distinctness in the consumer (reference). Pinned by
@@ -88,8 +88,7 @@ not to do, why it fails, and what to do instead.
88
88
  `depth[i]`/`dominates(depth, aligned)` to decide climb/IDF.
89
89
  - **WHY:** They measure different things: global reach (minority discriminates,
90
90
  powers climb/pooling) vs weave-local depth with `MIN_WEAVE=2` (what the local
91
- cohort shares, powers CAST) (`AGENTS §2` Cross-cutting contracts — Two
92
- measures of commonality).
91
+ cohort shares, powers CAST) (`docs/architecture/commonality.md`).
93
92
  - **CORRECT:** Climb/attention uses corpus-global;
94
93
  `frame(i) ⇔ depth[i]>MIN_WEAVE ∧ dominates(depth[i],aligned)` for CAST. Pinned
95
94
  by `test/50-cast-analog-consensus-floor.test.mjs` and
@@ -142,3 +141,32 @@ not to do, why it fails, and what to do instead.
142
141
  `twoEndedSeat`); turns are API state in `mind/mind.ts`, not segmentation.
143
142
  Pinned by `test/59-fold-invariance.test.mjs` and
144
143
  `test/63-fold-invariants.test.mjs`.
144
+
145
+ ### 13. Capping a combinatorial explosion instead of budgeting it
146
+
147
+ - **WRONG:** Answer a combinatorial explosion with a geometry-derived limit — a
148
+ cap on the pairs a sweep enumerates, the continuations a hop may offer, the
149
+ candidates a scan probes. A derived limit is the right cutoff for a DECISION;
150
+ used as the answer to explosion it is a short-circuit.
151
+ - **WHY:** It stops the computation silently. Reach is lost, the capability that
152
+ depended on it goes with it, and no test fails, because the tests were written
153
+ against the capped behaviour. Capping and removing the cap are both wrong:
154
+ capping truncates, removing lets the cost run, and the two failure modes hide
155
+ each other.
156
+ - **CORRECT:** BUDGET it. The work is charged in the one currency
157
+ (`MICRO`/`STEP`/`CONCEPT`/`PASS`; `weight = moves + PASS·unaccounted`, see
158
+ `docs/architecture/cost-model.md`), the charge is visible in the meter and the
159
+ rationale, and the SEARCH decides whether the work is worth paying — so
160
+ inference is never locked by a limit and nothing is truncated in silence.
161
+ Where the work is mechanical rather than evidential — enumeration, scans,
162
+ sweeps — the answer is an algorithm whose cost is structural in the bytes it
163
+ is given, not a smaller cap.
164
+ - **THE IDEAL:** a universal **closure engine** — one law of closure, stated in
165
+ the quantities the machine already has (`leadsSomewhere`, the
166
+ exact-then-canonical identity, `accounted` bytes, the ladder, `hubBound`),
167
+ from which the reach of a gap, the offer of a hop, the depth of a join and the
168
+ scope of a substitution are CONSEQUENCES, not four separate decisions. Nothing
169
+ in this repository is that today.
170
+ - **THE STANDARD A CHANGE MUST MEET:** state which consequence it is, and show
171
+ it following from the law. A change that cannot be stated that way is not
172
+ ready.
@@ -3,7 +3,7 @@
3
3
  Four executable gates. Each: run the command, check what it guards, follow its
4
4
  §.
5
5
 
6
- ## 1 — Correctness (all 87 suites)
6
+ ## 1 — Correctness (all 90 suites)
7
7
 
8
8
  ```bash
9
9
  npm test
@@ -13,8 +13,8 @@ Guards honest silence, determinism, and every pinned contract. Silence:
13
13
  unrelated queries ground to nothing (`test/28`, `50`, `56`, `67`, `76`, `84`).
14
14
  Determinism: same seed + deposit order + query gives byte-identical answer
15
15
  (`test/20`). Every invariant is pinned — a simplification that fails a test is
16
- wrong until the test is shown wrong. §14–25 (pipeline), §8 (derived thresholds),
17
- AGENTS.md §2 invariants 1–5.
16
+ wrong until the test is shown wrong. §14–25 (pipeline), §64 (derived
17
+ thresholds), AGENTS.md §2 invariants 1–5.
18
18
 
19
19
  ## 2 — Work accounting (profiler)
20
20
 
@@ -27,8 +27,8 @@ Guards without trace: counters deterministic and diffable between runs; phases
27
27
  nest (not disjoint — `think` contains every mechanism phase); shared analyses
28
28
  charged to themselves, not to the first toucher; millisecond fields are
29
29
  non-deterministic hints only. With `--trace`, recognition idempotence still
30
- holds (`test/42`). `src/meter.ts`, `docs/architecture/meter.md`, §26, AGENTS.md
31
- §2 invariant meter/cost.
30
+ holds (`test/42`). `src/meter.ts`, `docs/architecture/meter.md`, §55,
31
+ `AGENTS.md` §6.
32
32
 
33
33
  ## 3 — Dependency footprint
34
34
 
@@ -38,7 +38,7 @@ node --test test/88-dependency-footprint.test.mjs
38
38
 
39
39
  Guards `dist/src` imports only `node:` + relative paths, and `package.json`
40
40
  declares no `dependencies` (examples use `devDependencies` lazily). The
41
- near-zero footprint is a product feature. AGENTS.md §6, §3 (store has one
41
+ near-zero footprint is a product feature. AGENTS.md §7, §3 (store has one
42
42
  runtime dep: `node:sqlite`).
43
43
 
44
44
  ## 4 — Fold invariance and sublinear scaling
@@ -53,4 +53,4 @@ positional; grid regression (14.3% survival) cannot pass. `14` — inference cos
53
53
  is sublinear in corpus size (power-law exponent ≪ 1) and constant-rate in input
54
54
  length; measured on independent disjoint corpora via log–log slope.
55
55
  `src/geometry.ts` (`contentLevels`), `docs/architecture/fold-contract.md` +
56
- `bounded-reads.md`, §10, §29.
56
+ `bounded-reads.md`, §59, §63.
@@ -4,8 +4,8 @@
4
4
  // the cache ceiling, the read budgets, the caps. A knob that describes ONE
5
5
  // CORPUS (which pairs of SmolSent, how many SODA dialogues, how long an Aya
6
6
  // field may be) belongs next to that corpus's adapter, together with the
7
- // evidence that fixed its default — see AGENTS.md §2.16: a comment carries the
8
- // constraint, and a constraint is only readable beside the code it constrains.
7
+ // evidence that fixed its default — a comment carries the constraint, and a
8
+ // constraint is only readable beside the code it constrains.
9
9
 
10
10
  import { join } from "node:path";
11
11
 
@@ -37,7 +37,7 @@ import { convertedParquetUnits } from "./converted-parquet.js";
37
37
  //
38
38
  // So it displaces some wrong answers and manufactures others, INCLUDING turning
39
39
  // a correct silence into a wrong answer — and honest silence is a stated
40
- // property of this engine (AGENTS §2.13). On the mixed-curriculum store the
40
+ // property of this engine (INVARIANTS.md). On the mixed-curriculum store the
41
41
  // same shape produced the fragment "nus" for "wake me up at nine am".
42
42
  //
43
43
  // That evidence is four probes on toy stores and is NOT conclusive; it is,
@@ -13,7 +13,7 @@
13
13
  //
14
14
  // THE ONLY THIRD-PARTY CODE IN THIS REPOSITORY IS BELOW, and it is LAZILY
15
15
  // LOADED. Sema itself imports nothing outside `node:` — that is a product
16
- // property, not an accident (AGENTS.md §6) — and this trainer is an EXAMPLE,
16
+ // property, not an accident (AGENTS.md §7) — and this trainer is an EXAMPLE,
17
17
  // not part of the library. hyparquet (+ its Snappy codec) is therefore a dev
18
18
  // dependency, and it is loaded by a dynamic import the first time a Parquet
19
19
  // corpus is actually read: a curriculum with no Parquet stage (SmolSent,
package/jsr.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "$schema": "https://jsr.io/schema/config-file.v1.json",
3
3
  "name": "@hviana/sema",
4
- "version": "0.8.0",
4
+ "version": "0.8.2",
5
5
  "exports": "./src/index.ts"
6
6
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@hviana/sema",
3
- "version": "0.8.0",
3
+ "version": "0.8.2",
4
4
  "description": "Sema: a non-parametric, instance-based reasoning system.",
5
5
  "repository": {
6
6
  "type": "git",
package/src/config.ts CHANGED
@@ -116,6 +116,23 @@ export interface MindConfig {
116
116
  seed: number;
117
117
  recallQueryK: number;
118
118
  haloQueryK: number;
119
+ /** Corpus reading (see src/mind/corpus.ts): results per call, resolved
120
+ * nodes climbed from, contexts requested per climb, probes used to stride
121
+ * the id space when browsing, bytes of each side a preview keeps, and the
122
+ * smallest deposited note browsing will show. Capacities and budgets only —
123
+ * the one material floor (a resolved node must account for W bytes) is
124
+ * derived from the geometry, not declared here. */
125
+ corpusLimitMax: number;
126
+ corpusClimbs: number;
127
+ corpusContextsPerClimb: number;
128
+ corpusSampleProbes: number;
129
+ corpusPreviewBytes: number;
130
+ corpusSampleFloorBytes: number;
131
+ /** Items one rationale step may ITEMISE (the whole field is still counted in
132
+ * the step's note). A capacity of the rationale, not of recall: sharing
133
+ * `recallQueryK` meant `new Mind({recallQueryK: 100000})` un-bounded the very
134
+ * payload the bound exists for (found by an adversarial review). */
135
+ rationaleSampleK: number;
119
136
  normalizeEpsilon: number;
120
137
  cosineEpsilon: number;
121
138
 
@@ -131,6 +148,13 @@ export const DEFAULT_CONFIG: MindConfig = {
131
148
  seed: 42,
132
149
  recallQueryK: 12,
133
150
  haloQueryK: 12,
151
+ rationaleSampleK: 12,
152
+ corpusLimitMax: 24,
153
+ corpusClimbs: 24,
154
+ corpusContextsPerClimb: 6,
155
+ corpusSampleProbes: 6000,
156
+ corpusPreviewBytes: 220,
157
+ corpusSampleFloorBytes: 12,
134
158
  normalizeEpsilon: 1e-12,
135
159
  cosineEpsilon: 1e-12,
136
160
  alu: {
@@ -172,6 +196,17 @@ export function resolveConfig(opts: Partial<MindConfig> = {}): MindConfig {
172
196
  seed: opts.seed ?? DEFAULT_CONFIG.seed,
173
197
  recallQueryK: opts.recallQueryK ?? DEFAULT_CONFIG.recallQueryK,
174
198
  haloQueryK: opts.haloQueryK ?? DEFAULT_CONFIG.haloQueryK,
199
+ rationaleSampleK: opts.rationaleSampleK ?? DEFAULT_CONFIG.rationaleSampleK,
200
+ corpusLimitMax: opts.corpusLimitMax ?? DEFAULT_CONFIG.corpusLimitMax,
201
+ corpusClimbs: opts.corpusClimbs ?? DEFAULT_CONFIG.corpusClimbs,
202
+ corpusContextsPerClimb: opts.corpusContextsPerClimb ??
203
+ DEFAULT_CONFIG.corpusContextsPerClimb,
204
+ corpusSampleProbes: opts.corpusSampleProbes ??
205
+ DEFAULT_CONFIG.corpusSampleProbes,
206
+ corpusPreviewBytes: opts.corpusPreviewBytes ??
207
+ DEFAULT_CONFIG.corpusPreviewBytes,
208
+ corpusSampleFloorBytes: opts.corpusSampleFloorBytes ??
209
+ DEFAULT_CONFIG.corpusSampleFloorBytes,
175
210
  normalizeEpsilon: opts.normalizeEpsilon ?? DEFAULT_CONFIG.normalizeEpsilon,
176
211
  cosineEpsilon: opts.cosineEpsilon ?? DEFAULT_CONFIG.cosineEpsilon,
177
212
  alu: {