@hviana/sema 0.8.0 → 0.8.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. package/AGENTS.md +22 -1
  2. package/DATASETS.md +1 -1
  3. package/dist/example/train_base/config.js +2 -2
  4. package/dist/example/train_base/corpora/massive.js +1 -1
  5. package/dist/example/train_base/readers.js +1 -1
  6. package/dist/src/geometry.d.ts +10 -10
  7. package/dist/src/geometry.js +25 -24
  8. package/dist/src/meter.d.ts +4 -12
  9. package/dist/src/meter.js +14 -14
  10. package/dist/src/mind/attention.js +12 -12
  11. package/dist/src/mind/bridge.d.ts +8 -8
  12. package/dist/src/mind/bridge.js +33 -32
  13. package/dist/src/mind/graph-search.d.ts +0 -8
  14. package/dist/src/mind/graph-search.js +9 -8
  15. package/dist/src/mind/junction.d.ts +1 -1
  16. package/dist/src/mind/junction.js +8 -8
  17. package/dist/src/mind/learning.js +36 -35
  18. package/dist/src/mind/match.js +14 -13
  19. package/dist/src/mind/mechanisms/cover.js +13 -12
  20. package/dist/src/mind/mechanisms/prefix-completion.js +24 -24
  21. package/dist/src/mind/mechanisms/recall.js +38 -40
  22. package/dist/src/mind/mechanisms/reference.js +16 -16
  23. package/dist/src/mind/mind.d.ts +6 -7
  24. package/dist/src/mind/pipeline-mechanism.d.ts +10 -8
  25. package/dist/src/mind/pipeline-mechanism.js +25 -21
  26. package/dist/src/mind/pipeline.d.ts +9 -9
  27. package/dist/src/mind/pipeline.js +24 -23
  28. package/dist/src/mind/primitives.d.ts +5 -5
  29. package/dist/src/mind/primitives.js +5 -5
  30. package/dist/src/mind/recognition.d.ts +14 -13
  31. package/dist/src/mind/recognition.js +23 -23
  32. package/dist/src/mind/resonance.js +21 -21
  33. package/dist/src/mind/traverse.d.ts +54 -52
  34. package/dist/src/mind/traverse.js +74 -72
  35. package/dist/src/mind/types.d.ts +4 -4
  36. package/dist/src/store.d.ts +12 -12
  37. package/dist/src/store.js +12 -12
  38. package/docs/INDEX.md +2 -2
  39. package/docs/architecture/exact-vs-approximate.md +2 -1
  40. package/docs/architecture/fold-contract.md +1 -1
  41. package/docs/failures/tempting-but-wrong.md +2 -3
  42. package/docs/harness/gates.md +7 -7
  43. package/example/train_base/config.ts +2 -2
  44. package/example/train_base/corpora/massive.ts +1 -1
  45. package/example/train_base/readers.ts +1 -1
  46. package/jsr.json +1 -1
  47. package/package.json +1 -1
  48. package/src/geometry.ts +25 -24
  49. package/src/meter.ts +14 -14
  50. package/src/mind/attention.ts +12 -12
  51. package/src/mind/bridge.ts +33 -32
  52. package/src/mind/graph-search.ts +9 -8
  53. package/src/mind/junction.ts +8 -8
  54. package/src/mind/learning.ts +36 -35
  55. package/src/mind/match.ts +20 -19
  56. package/src/mind/mechanisms/cover.ts +13 -12
  57. package/src/mind/mechanisms/prefix-completion.ts +24 -24
  58. package/src/mind/mechanisms/recall.ts +38 -40
  59. package/src/mind/mechanisms/reference.ts +16 -16
  60. package/src/mind/mind.ts +6 -7
  61. package/src/mind/pipeline-mechanism.ts +25 -21
  62. package/src/mind/pipeline.ts +33 -32
  63. package/src/mind/primitives.ts +5 -5
  64. package/src/mind/recognition.ts +23 -23
  65. package/src/mind/resonance.ts +21 -21
  66. package/src/mind/traverse.ts +74 -72
  67. package/src/mind/types.ts +4 -4
  68. package/src/store.ts +20 -20
  69. package/test/08-storage.test.mjs +1 -1
  70. package/test/35-prefix-edge.test.mjs +1 -1
  71. package/test/40-choosenext-scale-guard.test.mjs +16 -17
  72. package/test/56-bridge-identity-admission.test.mjs +6 -6
  73. package/test/70-prefix-completion.test.mjs +4 -3
  74. package/test/72-prefix-candidate-supply.test.mjs +3 -3
  75. package/test/73-scaffolding-only-bridge-abstains.test.mjs +6 -6
  76. package/test/75-multiturn-context-optimisation.test.mjs +5 -5
  77. package/test/84-composed-answer-honesty.test.mjs +5 -6
  78. package/test/88-dependency-footprint.test.mjs +1 -1
  79. package/test/89-completion-recursion.test.mjs +17 -14
  80. package/test/90-connector-read-cap.test.mjs +10 -8
  81. package/test/93-regime-prediction.test.mjs +10 -10
  82. package/test/94-cross-region-budget.test.mjs +2 -2
  83. package/test/95-wide-resonance-removed.test.mjs +8 -7
  84. package/test/96-bytes-walk-termination.test.mjs +3 -3
@@ -9,11 +9,11 @@ import { cosine } from "../vec.js";
9
9
  import { gistOf, read } from "./primitives.js";
10
10
  import { canonicalWindows, leafIdPrefix, leafIdRun } from "./canonical.js";
11
11
  //
12
- // Budgeted on the same terms as the reach memo below (AGENTS §2.12): these
13
- // three maps are cleared on every write, but a long read-only session over a
14
- // large store converges on one entry per node per map with nothing to bound
15
- // it. Past the cap all three are dropped together and re-derived, costing
16
- // cold structural probes and never a wrong answer.
12
+ // Budgeted on the same terms as the reach memo below (caches.md): these three
13
+ // maps are cleared on every write, but a long read-only session over a large
14
+ // store converges on one entry per node per map with nothing to bound it. Past
15
+ // the cap all three are dropped together and re-derived, costing cold
16
+ // structural probes and never a wrong answer.
17
17
  const STRUCT_MEMO_MAX = 100_000;
18
18
  const structCaches = new WeakMap();
19
19
  // ── The shared ancestor-reach memo ──────────────────────────────────────
@@ -31,20 +31,21 @@ const structCaches = new WeakMap();
31
31
  // battery repeatedly reaches the same corpus scaffolding even when its
32
32
  // surface questions differ.
33
33
  //
34
- // Budgeted, not unbounded (AGENTS §2.12): past the cap the whole map is
35
- // dropped and re-derived, costing a cold climb and never a wrong answer.
34
+ // Budgeted, not unbounded (caches.md): past the cap the whole map is dropped
35
+ // and
36
+ // re-derived, costing a cold climb and never a wrong answer.
36
37
  const REACH_MEMO_MAX = 100_000;
37
38
  const reachCaches = new WeakMap();
38
39
  /** The reach memo this ask should use — see the note above.
39
40
  *
40
- * A TRACED response always gets a fresh, empty one. `AncestorReach`'s
41
- * `visited`/`maxDepth`/`saturation` fields are populated only when a trace
42
- * is attached, so an entry deposited by an untraced earlier turn would
43
- * silently black out the reach detail of a later traced one; and the trace's
44
- * reach payload is serialised by ITERATING this map, which must therefore
45
- * hold what THIS climb consulted, not the whole conversation's history.
46
- * Consistent with AGENTS §2.11: a traced response is a different machine —
47
- * never benchmark with a trace attached. */
41
+ * A TRACED response always gets a fresh, empty one. `AncestorReach`'s
42
+ * `visited`/`maxDepth`/`saturation` fields are populated only when a trace is
43
+ * attached, so an entry deposited by an untraced earlier turn would silently
44
+ * black out the reach detail of a later traced one; and the trace's reach
45
+ * payload is serialised by ITERATING this map, which must therefore hold what
46
+ * THIS climb consulted, not the whole conversation's history. Consistent with
47
+ * memoization.md: a traced response is a different machine — never benchmark
48
+ * with a trace attached. */
48
49
  export function sharedReachMemo(ctx) {
49
50
  if (ctx.trace !== null || ctx.climbMemo === null)
50
51
  return new Map();
@@ -426,13 +427,14 @@ export function bearsEdge(ctx, id) {
426
427
  return cachedHasNext(ctx, id, getStructCache(ctx));
427
428
  }
428
429
  /** Whether a node LEADS SOMEWHERE — it bears a continuation edge or a halo.
429
- * The admission predicate recognition filters sites with (HOW_IT_WORKS
430
- * §15.3): a form that leads nowhere contributes nothing to any derivation.
431
- * Runs once per candidate span on the recognition hot path — `hasNext` is
432
- * cached per response (the same flat-branch ids are probed across prefix
433
- * variants by canonicalChunkId). `hasHalo` is not cached: it's a single
434
- * indexed point probe per candidate, and the candidates that reach this
435
- * check have already been filtered by hasNext above in edgeAncestors. */
430
+ * The admission predicate recognition filters sites with (cover.md): a form
431
+ * that
432
+ * leads nowhere contributes nothing to any derivation. Runs once per candidate
433
+ * span on the recognition hot path — `hasNext` is cached per response (the same
434
+ * flat-branch ids are probed across prefix variants by canonicalChunkId).
435
+ * `hasHalo` is not cached: it's a single indexed point probe per candidate, and
436
+ * the candidates that reach this check have already been filtered by hasNext
437
+ * above in edgeAncestors. */
436
438
  export function leadsSomewhere(ctx, id) {
437
439
  const memo = getStructCache(ctx);
438
440
  if (cachedHasNext(ctx, id, memo))
@@ -476,10 +478,10 @@ function boundFor(contextCount) {
476
478
  return Math.ceil(Math.sqrt(Math.max(2, contextCount)));
477
479
  }
478
480
  /** Cap a candidate list at the hub bound √N (insertion order) — the ONE
479
- * fan-out convention every walk and disambiguation uses (see HOW_IT_WORKS
480
- * §8.6). A node connected to more than √N others is a hub whose individual
481
- * connections carry ~no discriminative information; materialising or scoring
482
- * them all would make single decisions scale with the corpus. */
481
+ * fan-out convention every walk and disambiguation uses (see bounded-reads.md).
482
+ * A node connected to more than √N others is a hub whose individual connections
483
+ * carry ~no discriminative information; materialising or scoring them all would
484
+ * make single decisions scale with the corpus. */
483
485
  export function hubCap(ctx, ids) {
484
486
  const bound = hubBound(ctx);
485
487
  return ids.length > bound ? ids.slice(0, bound) : ids;
@@ -512,16 +514,16 @@ export function contains(ctx, ancestor, descendant) {
512
514
  * the EXACT half's veto on calling them synonyms.
513
515
  *
514
516
  * Halos measure company, and the strongest company any two forms can keep is
515
- * standing next to each other: a question and its answer co-occur in every
516
- * episode that taught the pair, so their halos SHOULD be similar, and on a
517
- * conversational store they are (measured on the CONV fixture: consecutive
518
- * turns at 0.809 against a 0.516 concept threshold). A gate reading halo
519
- * cosine alone therefore reads adjacency as synonymy and revoices an answer
520
- * in the words of the question it answers — "it hangs in madrid" spliced back
521
- * into "where is it kept now". The distributional layer cannot tell the two
522
- * relations apart, because to it they are the same observation; the exact
523
- * half can, for free, because it stored the edge. §4.1's division of labour
524
- * exactly: approximate proposes, exact decides.
517
+ * standing next to each other: a question and its answer co-occur in every
518
+ * episode that taught the pair, so their halos SHOULD be similar, and on a
519
+ * conversational store they are (measured on the CONV fixture: consecutive
520
+ * turns at 0.809 against a 0.516 concept threshold). A gate reading halo cosine
521
+ * alone therefore reads adjacency as synonymy and revoices an answer in the
522
+ * words of the question it answers — "it hangs in madrid" spliced back into
523
+ * "where is it kept now". The distributional layer cannot tell the two
524
+ * relations apart, because to it they are the same observation; the exact half
525
+ * can, for free, because it stored the edge. halo-sketch.md's division of
526
+ * labour exactly: approximate proposes, exact decides.
525
527
  *
526
528
  * Read LIMITed in both directions at the hub bound — a common continuation's
527
529
  * fan-in is corpus-sized, and no single decision may scale with it. */
@@ -645,19 +647,18 @@ export function chooseNext(ctx, id, guide) {
645
647
  // NO consensusFloor gate here (tried and reverted — see
646
648
  // test/40-choosenext-scale-guard.test.mjs): that floor is calibrated for
647
649
  // POOLED, IDF-weighted CLIMB VOTES (recallByResonance, commitVotes), where
648
- // each corroborating region contributes at most ln N and the floor grows
649
- // with N exactly as that per-region ceiling does (HOW_IT_WORKS.md §8.6).
650
- // `bestSupport` here is a different kind of quantity — a raw prevCount of
651
- // how many training contexts predicted ONE destination, bounded by how
652
- // often that specific fact was retold, never by corpus size N. Gating an
653
- // N-invariant count against an N-growing threshold guarantees failure
654
- // once N is large enough, discarding genuinely, structurally dominant
655
- // edges (observed: a fact corroborated 2-to-1-1-1 refused at N≈325K,
656
- // falling back to a noisy concept-hop). The loop above already IS the
657
- // "genuinely competing" test: a tie leaves first-inserted as the pick
658
- // (test/30's own pinned behaviour); a strict winner is real evidence
659
- // regardless of corpus scale. Matches HOW_IT_WORKS.md §25's own
660
- // chooseNext pseudocode, which has no such floor.
650
+ // each corroborating region contributes at most ln N and the floor grows with
651
+ // N exactly as that per-region ceiling does (thresholds.md). `bestSupport`
652
+ // here is a different kind of quantity — a raw prevCount of how many training
653
+ // contexts predicted ONE destination, bounded by how often that specific fact
654
+ // was retold, never by corpus size N. Gating an N-invariant count against an
655
+ // N-growing threshold guarantees failure once N is large enough, discarding
656
+ // genuinely, structurally dominant edges (observed: a fact corroborated
657
+ // 2-to-1-1-1 refused at N≈325K, falling back to a noisy concept-hop). The
658
+ // loop above already IS the "genuinely competing" test: a tie leaves
659
+ // first-inserted as the pick (test/30's own pinned behaviour); a strict
660
+ // winner is real evidence regardless of corpus scale. Matches `chooseNext`'s
661
+ // own pseudocode, which has no such floor.
661
662
  // Trace is built lazily — the filter + map below only execute when a
662
663
  // trace listener is attached, so the common (no-trace) path pays only
663
664
  // for the prevCount calls in the loop above, never for extra rItemShort
@@ -709,12 +710,13 @@ function rItemShort(ctx, id, role, score) {
709
710
  * W-window it spells is contained by more places than the hub bound allows,
710
711
  * i.e. the whole query is corpus-global scaffolding.
711
712
  *
712
- * WHAT IT IS FOR. Several mechanisms ground a query through the literal
713
- * spans it did NOT explain, and those spans are the whole of their evidence.
714
- * When every one of them is a hub, the query says nothing the corpus can be
715
- * held to, and grounding it means picking one of thousands of continuations
716
- * it gives no evidence for — a fabrication whatever the answer happens to be.
717
- * Answering with silence there is the honest degradation contract (§2.13).
713
+ * WHAT IT IS FOR. Several mechanisms ground a query through the literal spans
714
+ * it
715
+ * did NOT explain, and those spans are the whole of their evidence. When every
716
+ * one of them is a hub, the query says nothing the corpus can be held to, and
717
+ * grounding it means picking one of thousands of continuations it gives no
718
+ * evidence for — a fabrication whatever the answer happens to be. Answering
719
+ * with silence there is the honest degradation contract (INVARIANTS.md).
718
720
  *
719
721
  * MEASURED SEPARATION (trained store, hubBound 571) — this is categorical,
720
722
  * not marginal, and it is why the predicate lives here rather than being
@@ -731,11 +733,11 @@ function rItemShort(ctx, id, role, score) {
731
733
  * evidence and sit on the SAME side as the correct ones, so this predicate
732
734
  * is not what makes them silent and cannot be credited for them.
733
735
  *
734
- * NO NEW THRESHOLD (§2.2): `hubBound` is the √N reading of "hub" used
735
- * everywhere, and the containment read is clamped to it exactly as every
736
- * other fan-out read is (§2.8). A query with no stored window at all is NOT
737
- * scaffolding-only — it has no evidence either way, and its callers already
738
- * refuse it on their own terms. */
736
+ * NO NEW THRESHOLD (thresholds.md): `hubBound` is the √N reading of "hub" used
737
+ * everywhere, and the containment read is clamped to it exactly as every other
738
+ * fan-out read is (bounded-reads.md). A query with no stored window at all is
739
+ * NOT scaffolding-only — it has no evidence either way, and its callers already
740
+ * refuse it on their own terms. */
739
741
  export function allWindowsAreScaffolding(ctx, query) {
740
742
  const W = ctx.space.maxGroup;
741
743
  const bound = hubBound(ctx);
@@ -786,19 +788,19 @@ export function allWindowsAreScaffolding(ctx, query) {
786
788
  * by climbing containment then parents. Nothing is added to the write side;
787
789
  * this reads an index training already built.
788
790
  *
789
- * BOUNDED (§2.8), AND WITH NO NEW THRESHOLD. The window whose containment is
790
- * SMALLEST carries the most evidence, and one saturated at `hubBound` carries
791
- * none — that is the same √N reading of "hub" the rest of the mind uses, not
792
- * a tuned knob. The upward walk spends a budget of `hubBound` nodes and
793
- * fans out by W, so a hub query enumerates nothing and the caller stays
794
- * silent rather than guessing (§2.13). Measured on the trained store: the
795
- * photosynthesis form at a one-byte truncation picks a window with 52
796
- * containers, visits 446 nodes, and yields exactly ONE candidate that
797
- * survives the caller's byte compare — the form itself.
791
+ * BOUNDED (bounded-reads.md), AND WITH NO NEW THRESHOLD. The window whose
792
+ * containment is SMALLEST carries the most evidence, and one saturated at
793
+ * `hubBound` carries none — that is the same √N reading of "hub" the rest of
794
+ * the mind uses, not a tuned knob. The upward walk spends a budget of
795
+ * `hubBound` nodes and fans out by W, so a hub query enumerates nothing and the
796
+ * caller stays silent rather than guessing (INVARIANTS.md). Measured on the
797
+ * trained store: the photosynthesis form at a one-byte truncation picks a
798
+ * window with 52 containers, visits 446 nodes, and yields exactly ONE candidate
799
+ * that survives the caller's byte compare — the form itself.
798
800
  *
799
- * These are PROPOSALS only. Every candidate still faces the byte-exact
800
- * prefix compare and all three guards below, so a wrong proposal costs one
801
- * bounded read and can never be voiced (§2.3). */
801
+ * These are PROPOSALS only. Every candidate still faces the byte-exact prefix
802
+ * compare and all three guards below, so a wrong proposal costs one bounded
803
+ * read and can never be voiced (exact-vs-approximate.md). */
802
804
  export function formsOpenedBy(ctx, query) {
803
805
  const store = ctx.store;
804
806
  const W = ctx.space.maxGroup;
@@ -250,10 +250,10 @@ export type AItem = {
250
250
  export interface MindContext extends GraphSearchHost {
251
251
  store: Store;
252
252
  /** The work accumulator for the inference call in flight, or null when
253
- * nothing is profiling — see src/meter.ts. WRITE-ONLY from the engine's
254
- * point of view: no inference decision may read a counter, or the
255
- * determinism contract (AGENTS §2.1) is gone. Every call site is
256
- * `ctx.meter?.x++`, so an unprofiled response allocates nothing. */
253
+ * nothing is profiling — see src/meter.ts. WRITE-ONLY from the engine's point
254
+ * of view: no inference decision may read a counter, or determinism is gone
255
+ * (determinism.md). Every call site is `ctx.meter?.x++`, so an unprofiled
256
+ * response allocates nothing. */
257
257
  meter: Meter | null;
258
258
  space: Space;
259
259
  alphabet: Alphabet;
@@ -701,18 +701,18 @@ export declare abstract class AbstractStore implements Store {
701
701
  * remainders must fit the budget. Scattered differences leave a wide
702
702
  * middle and are rejected.
703
703
  *
704
- * Every read here is CAPPED (§2.8). It used to open with
705
- * `bytesPrefix(k, Number.MAX_SAFE_INTEGER)` — the ALL sentinel, i.e. the
706
- * full materialising `bytes()` read — on the deposit hot path, and only
707
- * then compare lengths. So a candidate the length test was about to reject
708
- * had already been reconstructed byte for byte. The LENGTHS decide first
709
- * instead, from the `contentLen` memo the interning order has already built
710
- * bottom-up, and the target's length is itself read under a cap: a target
711
- * longer than `la + W` is rejected without touching one of its bytes.
712
- * Same semantics — the old capped `b` read would have produced
713
- * `a.length + W + 1` here and failed the very same test — strictly fewer
714
- * byte reads. The `+ 1` on each byte cap keeps `_prefix`'s
715
- * "complete reconstruction" test true, so the results still cache. */
704
+ * Every read here is CAPPED (bounded-reads.md). It used to open with
705
+ * `bytesPrefix(k, Number.MAX_SAFE_INTEGER)` — the ALL sentinel, i.e. the full
706
+ * materialising `bytes()` read — on the deposit hot path, and only then
707
+ * compare lengths. So a candidate the length test was about to reject had
708
+ * already been reconstructed byte for byte. The LENGTHS decide first instead,
709
+ * from the `contentLen` memo the interning order has already built bottom-up,
710
+ * and the target's length is itself read under a cap: a target longer than
711
+ * `la + W` is rejected without touching one of its bytes. Same semantics —
712
+ * the old capped `b` read would have produced `a.length + W + 1` here and
713
+ * failed the very same test — strictly fewer byte reads. The `+ 1` on each
714
+ * byte cap keeps `_prefix`'s "complete reconstruction" test true, so the
715
+ * results still cache. */
716
716
  private differsByOneWindow;
717
717
  putLeaf(bytes: Uint8Array, gist: Vec): Promise<NodeId>;
718
718
  putBranch(kids: NodeId[], gist: Vec): Promise<NodeId>;
package/dist/src/store.js CHANGED
@@ -1235,18 +1235,18 @@ export class AbstractStore {
1235
1235
  * remainders must fit the budget. Scattered differences leave a wide
1236
1236
  * middle and are rejected.
1237
1237
  *
1238
- * Every read here is CAPPED (§2.8). It used to open with
1239
- * `bytesPrefix(k, Number.MAX_SAFE_INTEGER)` — the ALL sentinel, i.e. the
1240
- * full materialising `bytes()` read — on the deposit hot path, and only
1241
- * then compare lengths. So a candidate the length test was about to reject
1242
- * had already been reconstructed byte for byte. The LENGTHS decide first
1243
- * instead, from the `contentLen` memo the interning order has already built
1244
- * bottom-up, and the target's length is itself read under a cap: a target
1245
- * longer than `la + W` is rejected without touching one of its bytes.
1246
- * Same semantics — the old capped `b` read would have produced
1247
- * `a.length + W + 1` here and failed the very same test — strictly fewer
1248
- * byte reads. The `+ 1` on each byte cap keeps `_prefix`'s
1249
- * "complete reconstruction" test true, so the results still cache. */
1238
+ * Every read here is CAPPED (bounded-reads.md). It used to open with
1239
+ * `bytesPrefix(k, Number.MAX_SAFE_INTEGER)` — the ALL sentinel, i.e. the full
1240
+ * materialising `bytes()` read — on the deposit hot path, and only then
1241
+ * compare lengths. So a candidate the length test was about to reject had
1242
+ * already been reconstructed byte for byte. The LENGTHS decide first instead,
1243
+ * from the `contentLen` memo the interning order has already built bottom-up,
1244
+ * and the target's length is itself read under a cap: a target longer than
1245
+ * `la + W` is rejected without touching one of its bytes. Same semantics —
1246
+ * the old capped `b` read would have produced `a.length + W + 1` here and
1247
+ * failed the very same test — strictly fewer byte reads. The `+ 1` on each
1248
+ * byte cap keeps `_prefix`'s "complete reconstruction" test true, so the
1249
+ * results still cache. */
1250
1250
  differsByOneWindow(kids, targetId, W) {
1251
1251
  const lens = kids.map((k) => this.contentLen(k));
1252
1252
  let la = 0;
package/docs/INDEX.md CHANGED
@@ -1,8 +1,8 @@
1
1
  # Sema Documentation Index
2
2
 
3
3
  Sema is a single system stated three ways: the law lives in `docs/architecture/`
4
- (what holds), the prescription in `AGENTS.md` bootloader (what to do and where),
5
- and the proof in `test/` (pins that fail when the law is broken).
4
+ (what holds), the prescription in `AGENTS.md` (what to do and where), and the
5
+ proof in `test/` (pins that fail when the law is broken).
6
6
 
7
7
  ## Routing — what to read for each task
8
8
 
@@ -44,4 +44,5 @@ Two rules in `attention.ts` encode "exact decides" and must not be flattened:
44
44
 
45
45
  Add a tier to the shared family in `mind/match.ts` with a derived gate
46
46
  (`geometry.ts`), never a private `score >= k` check. A new mechanism is a
47
- `(matcher, direction, gate)` configuration over that family (§2.5).
47
+ `(matcher, direction, gate)` configuration over that family
48
+ (`docs/architecture/match-project.md`).
@@ -84,4 +84,4 @@ any change here.
84
84
 
85
85
  See:
86
86
  `src/geometry.ts:contentLevels`/`contentBoundaries`/`contentFoldIncremental`/`stablePrefixFold`;
87
- `AGENTS.md` bootloader invariants.
87
+ `AGENTS.md` §2 invariants.
@@ -76,7 +76,7 @@ not to do, why it fails, and what to do instead.
76
76
  - **WHY:** `frameSlots` reports (contracted gaps tagged
77
77
  substitution/insertion/deletion); `carriesFillers` judges;
78
78
  `Precomputed.frames` inventories — elects nothing (`AGENTS §2` Cross-cutting
79
- contracts — `docs/architecture/factored-machinery.md` § Frame reading).
79
+ contracts — `docs/architecture/match-project.md` § Frame reading).
80
80
  - **CORRECT:** Report everything in the shared layer; apply
81
81
  `substituteAll(contA, fillersA→fillersB)==contB` and
82
82
  frame-dominance/`W`-reach/distinctness in the consumer (reference). Pinned by
@@ -88,8 +88,7 @@ not to do, why it fails, and what to do instead.
88
88
  `depth[i]`/`dominates(depth, aligned)` to decide climb/IDF.
89
89
  - **WHY:** They measure different things: global reach (minority discriminates,
90
90
  powers climb/pooling) vs weave-local depth with `MIN_WEAVE=2` (what the local
91
- cohort shares, powers CAST) (`AGENTS §2` Cross-cutting contracts — Two
92
- measures of commonality).
91
+ cohort shares, powers CAST) (`docs/architecture/commonality.md`).
93
92
  - **CORRECT:** Climb/attention uses corpus-global;
94
93
  `frame(i) ⇔ depth[i]>MIN_WEAVE ∧ dominates(depth[i],aligned)` for CAST. Pinned
95
94
  by `test/50-cast-analog-consensus-floor.test.mjs` and
@@ -3,7 +3,7 @@
3
3
  Four executable gates. Each: run the command, check what it guards, follow its
4
4
  §.
5
5
 
6
- ## 1 — Correctness (all 87 suites)
6
+ ## 1 — Correctness (all 90 suites)
7
7
 
8
8
  ```bash
9
9
  npm test
@@ -13,8 +13,8 @@ Guards honest silence, determinism, and every pinned contract. Silence:
13
13
  unrelated queries ground to nothing (`test/28`, `50`, `56`, `67`, `76`, `84`).
14
14
  Determinism: same seed + deposit order + query gives byte-identical answer
15
15
  (`test/20`). Every invariant is pinned — a simplification that fails a test is
16
- wrong until the test is shown wrong. §14–25 (pipeline), §8 (derived thresholds),
17
- AGENTS.md §2 invariants 1–5.
16
+ wrong until the test is shown wrong. §14–25 (pipeline), §64 (derived
17
+ thresholds), AGENTS.md §2 invariants 1–5.
18
18
 
19
19
  ## 2 — Work accounting (profiler)
20
20
 
@@ -27,8 +27,8 @@ Guards without trace: counters deterministic and diffable between runs; phases
27
27
  nest (not disjoint — `think` contains every mechanism phase); shared analyses
28
28
  charged to themselves, not to the first toucher; millisecond fields are
29
29
  non-deterministic hints only. With `--trace`, recognition idempotence still
30
- holds (`test/42`). `src/meter.ts`, `docs/architecture/meter.md`, §26, AGENTS.md
31
- §2 invariant meter/cost.
30
+ holds (`test/42`). `src/meter.ts`, `docs/architecture/meter.md`, §55,
31
+ `AGENTS.md` §6.
32
32
 
33
33
  ## 3 — Dependency footprint
34
34
 
@@ -38,7 +38,7 @@ node --test test/88-dependency-footprint.test.mjs
38
38
 
39
39
  Guards `dist/src` imports only `node:` + relative paths, and `package.json`
40
40
  declares no `dependencies` (examples use `devDependencies` lazily). The
41
- near-zero footprint is a product feature. AGENTS.md §6, §3 (store has one
41
+ near-zero footprint is a product feature. AGENTS.md §7, §3 (store has one
42
42
  runtime dep: `node:sqlite`).
43
43
 
44
44
  ## 4 — Fold invariance and sublinear scaling
@@ -53,4 +53,4 @@ positional; grid regression (14.3% survival) cannot pass. `14` — inference cos
53
53
  is sublinear in corpus size (power-law exponent ≪ 1) and constant-rate in input
54
54
  length; measured on independent disjoint corpora via log–log slope.
55
55
  `src/geometry.ts` (`contentLevels`), `docs/architecture/fold-contract.md` +
56
- `bounded-reads.md`, §10, §29.
56
+ `bounded-reads.md`, §59, §63.
@@ -4,8 +4,8 @@
4
4
  // the cache ceiling, the read budgets, the caps. A knob that describes ONE
5
5
  // CORPUS (which pairs of SmolSent, how many SODA dialogues, how long an Aya
6
6
  // field may be) belongs next to that corpus's adapter, together with the
7
- // evidence that fixed its default — see AGENTS.md §2.16: a comment carries the
8
- // constraint, and a constraint is only readable beside the code it constrains.
7
+ // evidence that fixed its default — a comment carries the constraint, and a
8
+ // constraint is only readable beside the code it constrains.
9
9
 
10
10
  import { join } from "node:path";
11
11
 
@@ -37,7 +37,7 @@ import { convertedParquetUnits } from "./converted-parquet.js";
37
37
  //
38
38
  // So it displaces some wrong answers and manufactures others, INCLUDING turning
39
39
  // a correct silence into a wrong answer — and honest silence is a stated
40
- // property of this engine (AGENTS §2.13). On the mixed-curriculum store the
40
+ // property of this engine (INVARIANTS.md). On the mixed-curriculum store the
41
41
  // same shape produced the fragment "nus" for "wake me up at nine am".
42
42
  //
43
43
  // That evidence is four probes on toy stores and is NOT conclusive; it is,
@@ -13,7 +13,7 @@
13
13
  //
14
14
  // THE ONLY THIRD-PARTY CODE IN THIS REPOSITORY IS BELOW, and it is LAZILY
15
15
  // LOADED. Sema itself imports nothing outside `node:` — that is a product
16
- // property, not an accident (AGENTS.md §6) — and this trainer is an EXAMPLE,
16
+ // property, not an accident (AGENTS.md §7) — and this trainer is an EXAMPLE,
17
17
  // not part of the library. hyparquet (+ its Snappy codec) is therefore a dev
18
18
  // dependency, and it is loaded by a dynamic import the first time a Parquet
19
19
  // corpus is actually read: a curriculum with no Parquet stage (SmolSent,
package/jsr.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "$schema": "https://jsr.io/schema/config-file.v1.json",
3
3
  "name": "@hviana/sema",
4
- "version": "0.8.0",
4
+ "version": "0.8.1",
5
5
  "exports": "./src/index.ts"
6
6
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@hviana/sema",
3
- "version": "0.8.0",
3
+ "version": "0.8.1",
4
4
  "description": "Sema: a non-parametric, instance-based reasoning system.",
5
5
  "repository": {
6
6
  "type": "git",
package/src/geometry.ts CHANGED
@@ -349,19 +349,20 @@ function bytesToLeaves(
349
349
  * sentences fall would be importing an assumption the architecture rejects.
350
350
  * Random binary must, and does, behave exactly like prose.
351
351
  *
352
- * Every constant is derived (§2.2): the cut mask is W, so a cut is offered once
353
- * per quantum of bytes — which, composed with the minimum below, puts the
354
- * expected segment at minLen + W − 1 ≈ 6 B rather than at W, deliberately (see
355
- * the refutation recorded at `cutRate` in {@link contentLevels}: a segment is
356
- * the flat PHRASE-scale unit the W-ary groups are built from, not a group of W
357
- * children, and forcing E[len] = W costs 15 tests). The minimum is W−1, `canonicalWindows`'s
358
- * straddle neighbour and the write side's own floor for a unit; and the maximum
359
- * is the KEYRING's seat count, because a segment folds as ONE flat node and
360
- * `fold` has exactly that many seats to bind children into. Capping there is
361
- * what keeps the fold light: a segment of 3..seats leaves is a single node,
362
- * where splitting it into W-groups plus a remainder would cost two or three
363
- * and the remainders barely share (measured: partial-arity nodes 504 → 3,590,
364
- * and total distinct nodes 8,142 → 9,712, when segments folded as [W][rest]). */
352
+ * Every constant is derived (thresholds.md): the cut mask is W, so a cut is
353
+ * offered once per quantum of bytes — which, composed with the minimum below,
354
+ * puts the expected segment at minLen + W − 1 ≈ 6 B rather than at W,
355
+ * deliberately (see the refutation recorded at `cutRate` in {@link
356
+ * contentLevels}: a segment is the flat PHRASE-scale unit the W-ary groups are
357
+ * built from, not a group of W children, and forcing E[len] = W costs 15
358
+ * tests). The minimum is W−1, `canonicalWindows`'s straddle neighbour and the
359
+ * write side's own floor for a unit; and the maximum is the KEYRING's seat
360
+ * count, because a segment folds as ONE flat node and `fold` has exactly that
361
+ * many seats to bind children into. Capping there is what keeps the fold light:
362
+ * a segment of 3..seats leaves is a single node, where splitting it into
363
+ * W-groups plus a remainder would cost two or three and the remainders barely
364
+ * share (measured: partial-arity nodes 504 → 3,590, and total distinct nodes
365
+ * 8,142 → 9,712, when segments folded as [W][rest]). */
365
366
  /** {@link contentBoundaries} plus, for each cut, its LEVEL — how deep in the
366
367
  * tree that cut reaches.
367
368
  *
@@ -622,16 +623,16 @@ export function knownPrefixLength(
622
623
  * correct boundary. Pass them through from `perceive`; the geometry
623
624
  * computes the stable prefix internally.
624
625
  *
625
- * `boundaries` is the CALLER-computed stable-prefix boundary set (§10.3):
626
- * strictly-increasing proper byte offsets, each the length of a prefix that
627
- * is already a stored whole-stream form. When given, the fold splits into
628
- * the segments between consecutive boundaries — each folded independently,
629
- * exactly as it folded when it was learned — and the segment roots join
630
- * LEFT-NESTED (((s₀·s₁)·s₂)…), so every learnt cumulative-context root
631
- * reappears as an identical subtree (and, by hash-consing, the very same
632
- * node) inside the grown stream. This is what lets a conversation's next
633
- * turn extend perception instead of refolding it: identical prefixes
634
- * produce identical subtrees regardless of what follows them. */
626
+ * `boundaries` is the CALLER-computed stable-prefix boundary set
627
+ * (fold-contract.md): strictly-increasing proper byte offsets, each the length
628
+ * of a prefix that is already a stored whole-stream form. When given, the fold
629
+ * splits into the segments between consecutive boundaries — each folded
630
+ * independently, exactly as it folded when it was learned — and the segment
631
+ * roots join LEFT-NESTED (((s₀·s₁)·s₂)…), so every learnt cumulative-context
632
+ * root reappears as an identical subtree (and, by hash-consing, the very same
633
+ * node) inside the grown stream. This is what lets a conversation's next turn
634
+ * extend perception instead of refolding it: identical prefixes produce
635
+ * identical subtrees regardless of what follows them. */
635
636
  export function bytesToTree(
636
637
  space: Space,
637
638
  alphabet: Alphabet,
@@ -962,7 +963,7 @@ function flatFold(
962
963
  return { tree: sema(gist, null, kids), len: n };
963
964
  }
964
965
 
965
- /** The stable-prefix segmented fold (§10.3). Each segment between
966
+ /* * The stable-prefix segmented fold (fold-contract.md). Each segment between
966
967
  * consecutive boundaries folds PLAINLY and independently; segment roots
967
968
  * join left-nested, and only the final root is normalized (the linear-fold
968
969
  * contract: one normalize per perception). A segment's own inner splits
package/src/meter.ts CHANGED
@@ -8,8 +8,8 @@
8
8
  // Four contracts, all load-bearing:
9
9
  //
10
10
  // 1. NEVER READ BY INFERENCE. No counter may reach a decision, a threshold,
11
- // or an ordering. Determinism (AGENTS §2.1) survives only because the
12
- // meter is write-only from the engine's point of view.
11
+ // or an ordering. Determinism (determinism.md) survives
12
+ // only because the meter is write-only from the engine's point of view.
13
13
  // 2. OFF BY DEFAULT, AND FREE WHEN OFF. Every call site is `meter?.x++` on
14
14
  // a null field. Nothing allocates, nothing is keyed, nothing is timed
15
15
  // unless a Meter is attached (`new Mind({ profile: true })`).
@@ -79,9 +79,9 @@ export class Meter {
79
79
  nodeRecords = 0;
80
80
  /** `store.bytes` / `store.bytesPrefix` — one reconstruction request. */
81
81
  byteReads = 0;
82
- /** Bytes actually handed back by those reads — the real I/O volume, and
83
- * the number that exposes an unbounded read (AGENTS §2.8) that a call
84
- * count alone hides. */
82
+ /* * Bytes actually handed back by those reads — the real I/O volume, and the
83
+ * number that exposes an unbounded read (bounded-reads.md) that a call count
84
+ * alone hides. */
85
85
  bytesRead = 0;
86
86
  /** `store.contentLen`. */
87
87
  lenReads = 0;
@@ -178,10 +178,10 @@ export class Meter {
178
178
  junctionPops = 0;
179
179
  /** Ascents that ended by EXHAUSTING the expansion budget rather than by
180
180
  * deciding — the walk abstained and the caller silently fell through to a
181
- * lower tier of the ladder (§2.13: a degradation nothing else reports).
182
- * It rises the moment a SHARED budget is drained by an earlier walk, which
183
- * is what makes "this tier answered nothing" distinguishable from "this
184
- * tier never got to look". */
181
+ * lower tier of the ladder — honest degradation, and nothing else reports it
182
+ * (INVARIANTS.md). It rises the moment a SHARED budget is drained by an
183
+ * earlier walk, which is what makes "this tier answered nothing"
184
+ * distinguishable from "this tier never got to look". */
185
185
  junctionBudgetExhausted = 0;
186
186
  /** Arbitrary byte spans whose distributional company was VSA-bundled from
187
187
  * existing episode halos. */
@@ -238,11 +238,11 @@ export class Meter {
238
238
  }
239
239
  }
240
240
 
241
- /** Time one SYNCHRONOUS phase. The sync/async seam (§2.10) is a real
242
- * contract — perception, recognition and the graph search are synchronous —
243
- * so a synchronous layer must not be wrapped in `time`'s promise just to be
244
- * measured: that would make the profiled path await where the unprofiled
245
- * one does not, and a meter never changes what a layer computes. */
241
+ /* * Time one SYNCHRONOUS phase. The sync/async seam is a real contract
242
+ * (meter.md) — perception, recognition and the graph search are synchronous —
243
+ * so a synchronous layer must not be wrapped in `time`'s promise just to be
244
+ * measured: that would make the profiled path await where the unprofiled one
245
+ * does not, and a meter never changes what a layer computes. */
246
246
  timeSync<T>(phase: string, fn: () => T): T {
247
247
  const before = this.snapshot();
248
248
  const t = performance.now();
@@ -1941,10 +1941,10 @@ export function canonicalChunkId(
1941
1941
  // CAST lost a point of attention it needed (test/29 D1/D2).
1942
1942
  //
1943
1943
  // So scan every offset and prefer an anchor that still discriminates: not
1944
- // saturated, and among those the one reaching the FEWEST contexts (§2.7,
1945
- // corpus-global). Only when every window in the region saturates does the
1946
- // old generalising choice stand — there is then no discriminative anchor to
1947
- // find, and abstaining is the honest outcome.
1944
+ // saturated, and among those the one reaching the FEWEST contexts
1945
+ // (commonality.md, corpus-global). Only when every window in the region
1946
+ // saturates does the old generalising choice stand — there is then no
1947
+ // discriminative anchor to find, and abstaining is the honest outcome.
1948
1948
  let discId: number | null = null;
1949
1949
  let discReached = Infinity;
1950
1950
  let fallback: number | null = null;
@@ -2649,16 +2649,16 @@ async function crossRegionVotes(
2649
2649
  const consumed = new Set<number>();
2650
2650
  let probes = 0;
2651
2651
  // When atoms themselves are hubs (atomIsHub — a single byte reaches ≥ √N
2652
- // contexts, §2.8's own predicate), the corpus is large enough that the
2653
- // cross-region junction walks are dominated by the drift through common
2654
- // content's ancestry. Each of k candidate pairs otherwise spends its own
2655
- // √N·W budget (profiled: 160,210 junction pops, 31% of think at
2656
- // N = 325,608), and a cumulative dialogue multiplies bounded work into tens
2657
- // of seconds. The structural walk is therefore given ONE k·W allowance per
2652
+ // contexts, bounded-reads.md's own predicate), the corpus is large enough
2653
+ // that the cross-region junction walks are dominated by the drift through
2654
+ // common content's ancestry. Each of k candidate pairs otherwise spends its
2655
+ // own √N·W budget (profiled: 160,210 junction pops, 31% of think at N =
2656
+ // 325,608), and a cumulative dialogue multiplies bounded work into tens of
2657
+ // seconds. The structural walk is therefore given ONE k·W allowance per
2658
2658
  // evidence tier, shared across every pair — k pairs × W phrase-scale levels,
2659
2659
  // the minimal exact check; a pair whose container is not reached within it
2660
- // falls through to the resonance tier (the ANN proposes what the shallow
2661
- // walk no longer exhaustively scans, §2.3).
2660
+ // falls through to the resonance tier (the ANN proposes what the shallow walk
2661
+ // no longer exhaustively scans, exact-vs-approximate.md).
2662
2662
  //
2663
2663
  // Below atomIsHub the store is small and atoms still discriminate, so the
2664
2664
  // walks keep exhaustive exact traversal (per-walk √N·W) — the shared budget