@hviana/sema 0.9.0 → 0.9.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (118) hide show
  1. package/AGENTS.md +7 -7
  2. package/dist/src/alu/src/index.d.ts +1 -1
  3. package/dist/src/alu/src/index.js +1 -1
  4. package/dist/src/alu/src/parser.js +2 -6
  5. package/dist/src/alu/src/resonance.d.ts +13 -0
  6. package/dist/src/alu/src/resonance.js +41 -0
  7. package/dist/src/alu/test/alu.test.js +39 -0
  8. package/dist/src/bytes.d.ts +6 -2
  9. package/dist/src/bytes.js +10 -4
  10. package/dist/src/canon.js +44 -0
  11. package/dist/src/geometry.d.ts +19 -1
  12. package/dist/src/geometry.js +125 -141
  13. package/dist/src/meter.d.ts +33 -0
  14. package/dist/src/meter.js +34 -1
  15. package/dist/src/mind/articulation.js +14 -1
  16. package/dist/src/mind/attention.d.ts +12 -0
  17. package/dist/src/mind/attention.js +44 -16
  18. package/dist/src/mind/bridge.js +3 -3
  19. package/dist/src/mind/derivation.d.ts +40 -0
  20. package/dist/src/mind/derivation.js +34 -0
  21. package/dist/src/mind/evidence.d.ts +24 -0
  22. package/dist/src/mind/evidence.js +90 -0
  23. package/dist/src/mind/graph-search.d.ts +89 -15
  24. package/dist/src/mind/graph-search.js +345 -174
  25. package/dist/src/mind/learning.js +1 -1
  26. package/dist/src/mind/mechanisms/cover.d.ts +19 -3
  27. package/dist/src/mind/mechanisms/cover.js +142 -61
  28. package/dist/src/mind/mechanisms/recall.js +10 -3
  29. package/dist/src/mind/mind.d.ts +6 -0
  30. package/dist/src/mind/mind.js +5 -2
  31. package/dist/src/mind/pipeline.d.ts +5 -1
  32. package/dist/src/mind/pipeline.js +220 -90
  33. package/dist/src/mind/primitives.d.ts +25 -5
  34. package/dist/src/mind/primitives.js +107 -44
  35. package/dist/src/mind/reasoning.d.ts +18 -4
  36. package/dist/src/mind/reasoning.js +487 -328
  37. package/dist/src/mind/recognition.js +29 -13
  38. package/dist/src/mind/resonance.js +1 -11
  39. package/dist/src/mind/traverse.d.ts +45 -5
  40. package/dist/src/mind/traverse.js +285 -8
  41. package/dist/src/mind/types.d.ts +16 -1
  42. package/dist/src/store-sqlite.d.ts +25 -0
  43. package/dist/src/store-sqlite.js +89 -1
  44. package/dist/src/store.d.ts +48 -4
  45. package/dist/src/store.js +86 -6
  46. package/docs/INDEX.md +20 -19
  47. package/docs/INVARIANTS.md +17 -16
  48. package/docs/architecture/bounded-reads.md +1 -1
  49. package/docs/architecture/caches.md +5 -4
  50. package/docs/architecture/closure.md +45 -5
  51. package/docs/architecture/cost-model.md +16 -0
  52. package/docs/architecture/evidence.md +113 -0
  53. package/docs/architecture/exact-vs-approximate.md +10 -9
  54. package/docs/architecture/factored-machinery.md +14 -13
  55. package/docs/architecture/fold-contract.md +51 -1
  56. package/docs/architecture/mechanism-market.md +21 -0
  57. package/docs/architecture/memoization.md +3 -3
  58. package/docs/architecture/meter.md +2 -1
  59. package/docs/architecture/saturation.md +12 -0
  60. package/docs/architecture/store.md +25 -2
  61. package/docs/failures/tempting-but-wrong.md +13 -2
  62. package/docs/harness/gates.md +12 -10
  63. package/docs/mechanisms/cover.md +23 -6
  64. package/jsr.json +1 -1
  65. package/package.json +1 -1
  66. package/src/alu/README.md +10 -2
  67. package/src/alu/src/index.ts +1 -0
  68. package/src/alu/src/parser.ts +6 -6
  69. package/src/alu/src/resonance.ts +42 -0
  70. package/src/alu/test/alu.test.ts +40 -0
  71. package/src/bytes.ts +13 -3
  72. package/src/canon.ts +40 -0
  73. package/src/geometry.ts +183 -154
  74. package/src/meter.ts +34 -1
  75. package/src/mind/articulation.ts +14 -2
  76. package/src/mind/attention.ts +47 -25
  77. package/src/mind/bridge.ts +3 -3
  78. package/src/mind/derivation.ts +77 -0
  79. package/src/mind/evidence.ts +107 -0
  80. package/src/mind/graph-search.ts +449 -221
  81. package/src/mind/learning.ts +1 -7
  82. package/src/mind/match.ts +1 -2
  83. package/src/mind/mechanisms/cast.ts +1 -2
  84. package/src/mind/mechanisms/cover.ts +207 -87
  85. package/src/mind/mechanisms/extraction.ts +1 -2
  86. package/src/mind/mechanisms/prefix-completion.ts +1 -1
  87. package/src/mind/mechanisms/recall.ts +17 -5
  88. package/src/mind/mechanisms/reference.ts +1 -1
  89. package/src/mind/mind.ts +9 -30
  90. package/src/mind/pipeline.ts +263 -104
  91. package/src/mind/primitives.ts +119 -43
  92. package/src/mind/reasoning.ts +611 -419
  93. package/src/mind/recognition.ts +24 -9
  94. package/src/mind/resonance.ts +2 -16
  95. package/src/mind/trace.ts +1 -1
  96. package/src/mind/traverse.ts +321 -8
  97. package/src/mind/types.ts +15 -11
  98. package/src/store-sqlite.ts +92 -1
  99. package/src/store.ts +113 -7
  100. package/test/105-derive-through-reports-its-refusal.test.mjs +8 -5
  101. package/test/106-the-join-fires.test.mjs +21 -0
  102. package/test/111-the-cover-assembly-is-counted.test.mjs +8 -5
  103. package/test/128-the-leads-somewhere-pair-agrees.test.mjs +18 -12
  104. package/test/136-the-two-named-limits.test.mjs +3 -2
  105. package/test/137-the-law-lives-once-and-below.test.mjs +21 -0
  106. package/test/148-exact-shortcuts-agree.test.mjs +188 -0
  107. package/test/149-the-closure-engine.test.mjs +138 -0
  108. package/test/150-the-join-is-output-sensitive.test.mjs +66 -0
  109. package/test/151-the-cover-pays-for-what-it-reaches.test.mjs +142 -0
  110. package/test/152-the-read-side-names-as-the-write-side.test.mjs +146 -0
  111. package/test/153-a-cheaper-bound-is-looked-at-first.test.mjs +155 -0
  112. package/test/154-the-question-names-the-step.test.mjs +281 -0
  113. package/test/24-generalization.test.mjs +32 -0
  114. package/test/36-bloom.test.mjs +53 -0
  115. package/test/37-cluster-dispersion-fusion.test.mjs +75 -0
  116. package/test/48-recognise-turn-connective.test.mjs +3 -2
  117. package/test/55-cost-meter.test.mjs +4 -4
  118. package/test/90-connector-read-cap.test.mjs +7 -7
@@ -0,0 +1,113 @@
1
+ # Witnessed Evidence — The Question Names the Step
2
+
3
+ > **Law:** a stored form is identified by the material at hand when every one of
4
+ > its bytes lies in a W-window that material holds — in any order, at any place,
5
+ > and wherever each piece of the material came from. A step the question did not
6
+ > name has to be paid for by material the question still owes.
7
+
8
+ ## The operation — `src/mind/evidence.ts`
9
+
10
+ `witness(form, indexes, W)` reads one form against a list of window indexes
11
+ (`windowIndex`). It is the order-free reading of correspondence, beside
12
+ `alignRuns` (which produces runs) and `junctionContainersFrom(…, unordered)`
13
+ (which finds containers). It is exact, deterministic and linear: one index per
14
+ source and one probe per window of the form. A window is credited to the LAST
15
+ source that holds it, so source 0 (the question) is credited only with what
16
+ nothing else at hand supplies. A form shorter than W is never witnessed.
17
+
18
+ ## Where the material comes from
19
+
20
+ The corpus records, for every continuation, the questions that establish it: its
21
+ predecessors. A 2Wiki fact `The father of Frederick II is Peter III of
22
+ Aragon.`
23
+ is established by `Frederick II` and by `Frederick II father`. The asker's
24
+ question rarely repeats either one byte for byte, but it often holds every byte
25
+ of one of them.
26
+
27
+ The derivation stands on more than the question. The node it is following is
28
+ material too. On the second hop of
29
+ `Where was the place of death of the
30
+ director of film Beat Girl?` the node is
31
+ `Edmond T. Gréville`, which the first hop reached and the asker never wrote. The
32
+ establishing question `Edmond T.
33
+ Gréville place of death` is held by neither the
34
+ question nor the first hop's fact, only by both. Measured over 5,236 held-out
35
+ 2WikiMultihopQA compositional questions, such a context is wholly witnessed by
36
+ the question alone 69 times, and by the question plus the first hop 2,153 times.
37
+
38
+ ## Its consumers
39
+
40
+ | Where | What it decides |
41
+ | --------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
42
+ | `chooseNext` (traverse.ts) | **The exact tier.** It names the continuation one of whose establishing contexts (other than the node) is witnessed by the question plus the node, with the question supplying at least one window the node does not. It ranks first among the readings; the distributional ladder decides only when nothing is named, or among continuations named equally. |
43
+ | `askedEvidence` (traverse.ts) | The question spans that named a pick. A mechanism projecting through the pick accounts for them, because evidence travels (mechanism-market.md). Recall's argument binding uses this. |
44
+ | `preConsumed` (pipeline.ts) | When a grounding does not declare `used`, what it spoke for is the forms inside its answer that the question already holds. The entity the answer added stays pivotable. |
45
+ | The walk (reasoning.ts) | A pivot is NAMED when one of its continuation's establishing contexts is witnessed by what of the question no product has said yet, plus the pivot. A named pivot MOVES. An unnamed one is offered only while the derivation still owes something, and the law then decides by carrying. |
46
+ | Cover sites (mechanisms/cover.ts) | A FRAGMENT is a form that sits inside other forms, has several continuations, and leaves at least one window of the question beyond it. It answers other questions, so it leads somewhere for this question only when the question names one of its continuations. |
47
+
48
+ A cover span made of nothing but scaffolding (every window a hub,
49
+ `scaffoldSpans`) is not accounted. The form recognised there, such as the song
50
+ `What` in the trained store, is one of thousands the bytes could name. This is
51
+ measured only on the trained store (`What country is Jerry Bock a citizen of?`
52
+ was won by `The performer of What is Melinda Marx.` glued onto the right fact).
53
+ A synthetic fixture could not reproduce that regime, because content addressing
54
+ folds every filler's `What` into a handful of shared nodes, so no suite test
55
+ pins this rule.
56
+
57
+ ## What the question owes
58
+
59
+ The derivation is born owing only its discriminative material. The bytes a
60
+ corpus-global scaffolding window reaches (`scaffoldExtents`, the same "hub"
61
+ reading as `allWindowsAreScaffolding` and the bridge's `explainedSpan`) are
62
+ nobody's debt. Otherwise a step could claim to pay `Who is the` by restating
63
+ `is`, which every fact holds. Pricing is untouched: the ladder still charges
64
+ every unexplained byte, because for a question made only of scaffolding,
65
+ covering those bytes is the evidence. Making scaffolding free in the market was
66
+ measured and refused, because it changed dialogue answers
67
+ (`How are you
68
+ today?`).
69
+
70
+ The hub reading behind `scaffoldExtents` is floored at `chainReach(W)`
71
+ containers. Inside one deposit's fold, a window is already contained by up to
72
+ that many chunks and branches, so on a store of a few facts the √N reading would
73
+ call every window frame. That count measures fold structure, not corpus
74
+ commonality (`test/22`'s two-fact chains are exactly that regime).
75
+
76
+ ## Bounds
77
+
78
+ The exact tier reads at most √N establishing contexts per decision, floored at
79
+ `chainReach(W)`. It asks the cheapest candidates first and compares scores
80
+ afterwards in the continuations' own order. It abstains, metered as
81
+ `askedReadsSaturated`, in two cases: the continuation read came back at the √N
82
+ cap, or the question holds no window the node lacks. One short prefix read
83
+ refuses most predecessors before a whole form is reconstructed. Picks are
84
+ memoized per node per question.
85
+
86
+ ## Measured
87
+
88
+ | Measure | Before | After |
89
+ | -------------------------------------------------------------------------------------- | ------ | ------ |
90
+ | 2Wiki held-out fixture (300 rows, deposited as `wiki2.ts` does), compositional correct | 13/133 | 47/133 |
91
+ | Same fixture, pivot steps | 3 | 90+ |
92
+ | Same fixture, inference correct | 8/37 | 6/37 |
93
+
94
+ The inference drop is real. Two hops of `father` from a question that says
95
+ `father` once, as in `paternal grandfather`, used to be reached by
96
+ over-extension and are no longer. The store holds no evidence that `grandfather`
97
+ composes `father` twice.
98
+
99
+ On the 31.7M-node store's 116-query battery, two answers became correct
100
+ (`What country is Jerry Bock a citizen of?` and
101
+ `What is the country of citizenship of Frederick II?`), two lost a junk
102
+ composition, three changed between wrong answers, and no correct answer was
103
+ lost. CPU was 150 s against two baseline runs of 139 s and 166 s, inside the
104
+ noise. The deterministic read counters rose about 20% (`bytesRead`,
105
+ `nodeRecords`), with ANN queries unchanged.
106
+
107
+ ## Pins
108
+
109
+ - `test/154` — witnessing semantics. The named continuation beats the
110
+ most-poured one. The second hop is named by the question plus the introduced
111
+ entity, whatever grounded the first hop. A question that names no further step
112
+ is not extended. A fragment voices none of its continuations unless the
113
+ question names one. Each assertion was verified by mutation.
@@ -1,4 +1,4 @@
1
- # Exact vs Approximate — The Law and Its Five Ladders
1
+ # Exact vs Approximate — The Law and Its Six Ladders
2
2
 
3
3
  Vector scores (`resonate` / `resonateHalo`) are RaBitQ **estimates**. They rank
4
4
  candidates and gate broad regions; they never decide identity. Identity is
@@ -15,17 +15,18 @@ thresholds gate breadth, not truth.
15
15
 
16
16
  ## Graded evidence ladders
17
17
 
18
- Five subsystems share one shape — **exact → distributional → geometric** — with
18
+ Six subsystems share one shape — **exact → distributional → geometric** — with
19
19
  earlier tiers strictly preferred. Never reorder tiers; never let an approximate
20
20
  tier override an exact one.
21
21
 
22
- | # | Site | Ladder (strong → weak) | File |
23
- | - | ------------------ | ------------------------------------------------------------------------------------------------------------------------ | ----------------------------------------- |
24
- | 1 | `resolve` | exact content-addressed fold → `canonResolve` (equivalence class, hash-then-verify) | `mind/primitives.ts` |
25
- | 2 | `locate` | exact bytes → halo role → gist | `mind/match.ts` |
26
- | 3 | `alignGraded` | literal W-gram runs → halo-matched sites + climb proposals (weave) | `mind/match.ts` / `pipeline-mechanism.ts` |
27
- | 4 | `bridge` | junction containers → edge → synonym → whole-gist | `mind/resonance.ts` |
28
- | 5 | `crossRegionVotes` | exact containers → single synonym → double → `structuralResonance` (synthetic gist, gated hardest — no byte containment) | `mind/attention.ts` |
22
+ | # | Site | Ladder (strong → weak) | File |
23
+ | - | ------------------ | --------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------- |
24
+ | 1 | `resolve` | exact content-addressed fold → `canonResolve` (equivalence class, hash-then-verify) | `mind/primitives.ts` |
25
+ | 2 | `locate` | exact bytes → halo role → gist | `mind/match.ts` |
26
+ | 3 | `alignGraded` | literal W-gram runs → halo-matched sites + climb proposals (weave) | `mind/match.ts` / `pipeline-mechanism.ts` |
27
+ | 4 | `bridge` | junction containers → edge → synonym → whole-gist | `mind/resonance.ts` |
28
+ | 5 | `crossRegionVotes` | exact containers → single synonym → double → `structuralResonance` (synthetic gist, gated hardest — no byte containment) | `mind/attention.ts` |
29
+ | 6 | `chooseNext` | an establishing context witnessed by the question ∪ the node (evidence.md) → distributional support (prevCount, poured mass) → first-inserted | `mind/traverse.ts` |
29
30
 
30
31
  ## Asymmetries (attention)
31
32
 
@@ -7,19 +7,20 @@ Siblings: `match-project.md`, `commonality.md`, `meter.md`.
7
7
 
8
8
  ## Single-definition contracts
9
9
 
10
- | Symbol | Defined in | One fact |
11
- | ------------------------------------------------------------- | ------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
12
- | `contentLevels` | `src/geometry.ts` | Single boundary rule: cuts + levels from one rolling hash pass; every segmentation reads it. |
13
- | `canonicalWindows` / `chainReach` / `leafIdRun` / `windowIds` | `src/mind/canonical.ts` | Write/read contract: training interns `W-1,W` windows, reading chains to `W²` and probes `W`-windows — drift silences recognition. |
14
- | `junction.ts` + `WalkCache` | `src/mind/junction.ts` | Shared junction ascent (parents + containers) with bounded `√N·W` walk; `WalkCache` memoizes capped reads/parents/containers per response; bridge and attention share it. |
15
- | `joinWithBridge` | `src/mind/resonance.ts` | One out-of-search assembly: `bridge(left,right)` or bare concat with `bridgeMiss` trace. |
16
- | `dismissedKnownContent` | `src/mind/bridge.ts` | Pure attestation: any unaccounted `W`-window that resolves as known content — shared gap guard for substitution and CAST. |
17
- | `sharedReachMemo` | `src/mind/traverse.ts` | One response-scoped `AncestorReach` memo (cleared on write and for traces); every `reachOf`/`edgeAncestors` consumer shares it. |
18
- | `guidedFirst` | `src/mind/traverse.ts` | Guided-or-first answer bytes: `guidedNext` else first-inserted edge (`LIMIT 1`). |
19
- | `leadsSomewhere` | `src/mind/traverse.ts` | Admission predicate: `hasNext` (cached) or `hasHalo`; sites that lead nowhere contribute no derivation. |
20
- | `isChunk` | `src/sema.ts` | `kids !== null && kids.every(k=>k.kids===null)` — smallest grouped unit; governs regions, seams, indexing. |
21
- | `twoEndedSeat` | `src/sema.ts` | One seat algebra: first half low seats, second half high seats; shared by perception, `fold`, and canonical folds. |
22
- | `closed`/`admissible`/`advance` | `src/mind/derivation.ts` | One admission for every derivation step and the readings every tier asks. |
10
+ | Symbol | Defined in | One fact |
11
+ | ------------------------------------------------------------- | ------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
12
+ | `contentLevels` | `src/geometry.ts` | Single boundary rule: cuts + levels from one rolling hash pass; every segmentation reads it. |
13
+ | `canonicalWindows` / `chainReach` / `leafIdRun` / `windowIds` | `src/mind/canonical.ts` | Write/read contract: training interns `W-1,W` windows, reading chains to `W²` and probes `W`-windows — drift silences recognition. |
14
+ | `junction.ts` + `WalkCache` | `src/mind/junction.ts` | Shared junction ascent (parents + containers) with bounded `√N·W` walk; `WalkCache` memoizes capped reads/parents/containers per response; bridge and attention share it. |
15
+ | `joinWithBridge` | `src/mind/resonance.ts` | One out-of-search assembly: `bridge(left,right)` or bare concat with `bridgeMiss` trace. |
16
+ | `dismissedKnownContent` | `src/mind/bridge.ts` | Pure attestation: any unaccounted `W`-window that resolves as known content — shared gap guard for substitution and CAST. |
17
+ | `sharedReachMemo` | `src/mind/traverse.ts` | One response-scoped `AncestorReach` memo (cleared on write and for traces); every `reachOf`/`edgeAncestors` consumer shares it. |
18
+ | `guidedFirst` | `src/mind/traverse.ts` | Guided-or-first answer bytes: `guidedNext` else first-inserted edge (`LIMIT 1`). |
19
+ | `leadsSomewhere` | `src/store.ts` (raw), `src/mind/traverse.ts` (memoised) | Admission predicate: `hasNext` or `hasHalo`, defined once on the store; traverse memoises the edge tier per response. |
20
+ | `isChunk` | `src/sema.ts` | `kids !== null && kids.every(k=>k.kids===null)` — smallest grouped unit; governs regions, seams, indexing. |
21
+ | `twoEndedSeat` | `src/sema.ts` | One seat algebra: first half low seats, second half high seats; shared by perception, `fold`, and canonical folds. |
22
+ | `closed`/`admissible`/`advance`/`closeOver` | `src/mind/derivation.ts` | One admission for every derivation step, the readings every tier asks, and the engine that walks the post-grounding layers. |
23
+ | `exactNode` | `src/mind/primitives.ts` | The exact content-addressed lookup (`resolve`, `canonResolve`): the identity fold (`contentIdentity`, geometry.ts) — a miss decided by the segment filter, a hit named bottom-up, no vectors. |
23
24
 
24
25
  ## Pins
25
26
 
@@ -56,6 +56,50 @@ drift without a type error. Levels are read from the hash the cut was accepted
56
56
  at — level `L` when `h` vanishes mod `W^(L+1)` — so level-`L` cuts nest inside
57
57
  level-`(L-1)` and expected span is `W^(L+1)` bytes.
58
58
 
59
+ ## One shape, two algebras
60
+
61
+ Which items group under which parent is decided by the cut levels, the keyring
62
+ and — inside an over-long row — each item's `itemKey` (eight raw gist
63
+ coordinates). `groupByLevel`/`foldSlice` are written ONCE over a fold algebra
64
+ (`join`, `key`) and run over two of them: the vector fold (`perceive`) and the
65
+ identity fold (`contentIdentity`), which names the node a stream folds to by
66
+ asking the store bottom-up and reads only the coordinates `itemKey` hashes,
67
+ lazily, with the same float32 additions in the same order. `exactNode` is the
68
+ identity fold, so resolving a span builds no D-dimensional gist and leaves
69
+ nothing in the perception memo (measured on the 31.7M-node store: one join query
70
+ went 185 s / 4.4 GB retained → 40 s / 81 MB, same answer). A grouping rule
71
+ written twice would be a write/read drift waiting to happen; written once, the
72
+ two folds cannot disagree about the tree.
73
+
74
+ ## Same tree is not enough — the read side names as the write side names
75
+
76
+ The store's `intern` names a branch by its children, and when they name none it
77
+ looks up the FLAT node over the same bytes and REUSES it (step 1b, "same bytes,
78
+ same node"). Every deposit interns flat nodes — its whole input and each
79
+ canonical window — so a later deposit's branch is often stored as an earlier
80
+ deposit's window (`ver` + `!` stored as the window `ver!`). The same tree is
81
+ then named by a node no child lookup can reach. `exactNode` and `foldTree` name
82
+ a branch exactly as `intern` does (`branchNaming`, `src/mind/primitives.ts`):
83
+ the children first, the flat node over the same bytes when they name nothing —
84
+ and an unnamed child does not settle it, since the write side minted that child
85
+ and still reached step 1b. Measured on the 31.7M-node store: 7 of 80 dialogue
86
+ turns asked verbatim had resolved to nothing and fell to the composition path
87
+ (26–41 s); they now resolve to their own context.
88
+
89
+ A name found only through the bytes is where the exact lookup used to MISS, and
90
+ a flat index entry is not a learnt structure, so `resolve` still asks the
91
+ canonical class there (`exactNaming`'s `byBytes`): it holds the learnt member
92
+ that leads somewhere, when there is one. The store holds such byte-only names
93
+ from an earlier deposit path (today's deposits always intern the structure
94
+ beside the flat copy), which is why `test/152` builds the state through the
95
+ store's write API.
96
+
97
+ One stored turn of the same set still does not resolve: its context was stored
98
+ with cuts at the ends of the conversation's EARLIER contexts (103 and 227 bytes
99
+ into a 257-byte context) — a boundary-imposed shape neither today's deposit path
100
+ nor the training-time one produces from these bytes, and that the read side
101
+ could reproduce only by guessing turn boundaries, which this contract forbids.
102
+
59
103
  ## Optional canonical capability
60
104
 
61
105
  `canonAdd`/`canonFind` (`src/store.ts` — `canonCount`/`eachContent`) is an
@@ -81,7 +125,13 @@ any change here.
81
125
  - `test/59` — shift invariance floors (content-defined cuts preserved over
82
126
  random binary and prose).
83
127
  - `test/63` — offset/W invariance and `contentLevels` distribution expectations.
128
+ - `test/148` — the identity fold names exactly what the vector fold names (every
129
+ sub-span of corpus and noise), and groups exactly as it does over long
130
+ low-entropy streams that force the `itemKey` split.
131
+ - `test/152` — a deposit stored through an earlier deposit's flat window
132
+ resolves to its own context, both folds name it, and it is answered on the
133
+ exact path.
84
134
 
85
135
  See:
86
- `src/geometry.ts:contentLevels`/`contentBoundaries`/`contentFoldIncremental`/`stablePrefixFold`;
136
+ `src/geometry.ts:contentLevels`/`contentBoundaries`/`contentFoldIncremental`/`stablePrefixFold`/`contentIdentity`;
87
137
  `AGENTS.md` §2 invariants.
@@ -67,6 +67,25 @@ Weight is one currency: `weight = moves + PASS · unaccountedBytes` where
67
67
  Cover runs first so a near-zero-cost computed span prunes the rest through the
68
68
  same mechanism — not a special case.
69
69
 
70
+ - **A cheaper bound is looked at first.** Before mechanism `m` first-touches
71
+ anything, every LATER mechanism whose floor grade is strictly below `m`'s runs
72
+ ahead of it (cheapest first); the lowest grade they reach is `bound`, and any
73
+ mechanism floored above `bound` is skipped (`meter.mechanismsBounded`). The
74
+ bound is learnt by calling `floor` with a `worthRunning` that refuses — the
75
+ investment discipline makes that free. The DECISION is the declared order's: a
76
+ run-ahead mechanism bounds the final grade whether or not the declared order
77
+ would have run it (if pruned, the incumbent already sat at or below its
78
+ floor); every candidate above `bound` loses to the winner, and every mechanism
79
+ floored at or below it meets the same run-or-prune decision, so `consider`
80
+ replays the same candidates in declared order. Equal floors are not skipped,
81
+ so an earlier mechanism keeps the tie it would win. Running ahead is never
82
+ extra work: only a mechanism floored at or below `p` can prune `p`, and each
83
+ such mechanism has already run or runs ahead of `p`. Measured on the
84
+ 31.7M-node store: a lowercased Persian turn (#83) went from 18.0 s to 1.2 s,
85
+ and #114 from 1.9 s to 0.7 s. In both, a grade-1 recall or prefix answer no
86
+ longer waits behind CAST's climb and weave. Of 42 composition-regime queries,
87
+ none changed its answer.
88
+
70
89
  - **Investment discipline.** `worthRunning` is passed _into_ `floor`. A floor
71
90
  that would first-touch an expensive shared analysis (`pre.attention()` climb,
72
91
  `pre.weave()`, `pre.resonance()`) checks `worthRunning(cheapestBound)` first
@@ -93,3 +112,5 @@ out so `PASS`-bridged bytes are still charged. `narrowDecision` and
93
112
 
94
113
  - `test/01-floor` — floor geometry.
95
114
  - `test/04-think` — decider, admissible pruning, investment discipline.
115
+ - `test/153` — run-ahead bounds: the composition market is skipped below CAST's
116
+ floor, and the decision equals a declared-order oracle.
@@ -47,9 +47,9 @@ conversation's persistent ones (content-keyed, cross-turn).
47
47
  | Memo | Key | Scope |
48
48
  | ------------------- | ----------------------------------------- | -------------------------------------------- |
49
49
  | `perceiveMemo` | `perceiveKey(bytes)` (latin1) | response / conversation |
50
- | `recogniseMemo` | `latin1Key(bytes)` | response / conversation |
51
- | `climbMemo` | `latin1Key(bytes)` | response / conversation |
52
- | `canonMemo` | `latin1Key(bytes)` | response (when `canon` set) |
50
+ | `recogniseMemo` | `latin1(bytes)` | response / conversation |
51
+ | `climbMemo` | `latin1(bytes)` | response / conversation |
52
+ | `canonMemo` | `latin1(bytes)` | response (when `canon` set) |
53
53
  | `_resolvedSubtrees` | `WeakMap<Sema, {id,len}>` (node identity) | response / conversation |
54
54
  | `_edgeChoice` | `Map<nodeId, pick>` | response only — **cleared** in `endResponse` |
55
55
  | `_gistCache` | `BoundedMap<nodeId, Vec>` 32 MB | **session-lifetime** (not per-response) |
@@ -3,7 +3,8 @@
3
3
  `src/meter.ts` is the one computational-usage accounting surface. It counts what
4
4
  inference _cost_ so a slow response can be attributed instead of guessed at. The
5
5
  rationale says why an answer was chosen; the meter says what it cost to choose
6
- it. Harness: `bench/profile-inference.mjs`.
6
+ it. Harness: the public path — `new Mind({ profile: true })`, then
7
+ `mind.lastCost` (`formatReport` / `sumReports`).
7
8
 
8
9
  ## Five contracts
9
10
 
@@ -48,6 +48,18 @@ full — identical to the unbounded climb. Work is `O(bound)` contexts times loc
48
48
  structure, never `O(N)`. Container seeding is streamed in `bound`-sized pages
49
49
  for the same reason.
50
50
 
51
+ Refuted tightening: **container fan-out as a hub** — deciding a containment seed
52
+ saturated when its first `containersSlice(bound+1)` page overflows, the reading
53
+ `junction.ts` applies to the same links. It conflates the PLACES a window occurs
54
+ with the CONTEXTS it reaches, and saturation is defined over contexts. Proved on
55
+ `test/49`'s fixture (N = 5, `sqrt(N)` = 3): `fran` sits in 4 containers yet its
56
+ full climb reaches 2 contexts — discriminative — and the rule called it
57
+ saturated, as it did `of f`, `fra`, `is`, `ranc`, `ance`. On the 31.7M-node
58
+ store no such window climbed unsaturated (18 composition queries; those climbs
59
+ were 38% of all visits), because at `sqrt(N)` = 1,560 a full page of containers
60
+ rarely converges on few contexts — but the rule is the same wrong reading at
61
+ both scales. The streamed seed stays.
62
+
51
63
  Trace records the first deciding stop as
52
64
  `SaturationStop { reason, node, observed, limit }` plus `visited`/`maxDepth`;
53
65
  absent when unsaturated or untraced.
@@ -27,7 +27,23 @@ a row. `has(id)` is `id < 0 || id < _nextId`.
27
27
  A branch whose kids are all leaves is **flat** — stored as raw bytes in `leaf`
28
28
  with an empty `kids` blob as marker (`flatKidsBytes`/`flatBytesKids`). Dedup
29
29
  probes hash then verify: `hashOf`→`h`→`LIMIT 1` fetch→byte compare (bloom
30
- negative filter first).
30
+ negative filter first). `flatBranchMayExist(bytes)` is the filter alone — its
31
+ `false` is exact, its `true` means "look it up" — for a caller that will verify
32
+ anyway: `exactNode` (primitives.ts) refuses a stream whose level-0 segment
33
+ cannot exist before naming any node, so a resolve miss costs hashing.
34
+ `findFlatBranch` memoizes HITS (`_flatKey`, keyed by the bytes themselves) and
35
+ never misses — the filter answers first, so a span that is not stored builds no
36
+ key, while the segments the identity fold names, asked by every span that
37
+ contains them, cost a map hit. `flatSpans(bytes)` returns a prober over ONE
38
+ buffer that answers exactly `findFlatBranch(bytes.subarray(s, e))`. `hashOf` is
39
+ FNV-1a, a left fold, so the prober extends the hash each start was last probed
40
+ at, and keeps one byte-short copy for the edge scan's trimmed retry. A span
41
+ scanner sweeping ends upward pays O(1) per probe instead of O(span).
42
+ Recognition's interior pass was hashing O(n·reach²) bytes: 251,660,406 for one
43
+ response on the 31.7M-node store (`test/153.4`).
44
+
45
+ `leadsSomewhere(id)` — `hasNext || hasHalo` — is the admission predicate's ONE
46
+ raw definition; `traverse.ts` memoises its edge tier per response.
31
47
 
32
48
  `bytes(id)`/`bytesPrefix(id, cap)` are shared with `BoundedMap` caches — callers
33
49
  must **never mutate** the returned buffer. `contentLen(id, cap)` walks with
@@ -43,7 +59,14 @@ window apart. Gists sit in `_pendingGist` (byte-budgeted `BoundedMap`);
43
59
  `batchSize` batches. Buffers flush on cadence, `commit()`, and close. Halo mass
44
60
  re-indexes geometrically (`mass<=4 || powerOfTwo`) and encodes 2-bit quantized.
45
61
  Canon index is optional: `canonAdd`/ `canonFind`/`canonCount` over 32-bit
46
- canonical hashes, caller verifies bytes.
62
+ canonical hashes, caller verifies bytes. The SQLite backend keeps a negative
63
+ filter over the canon hashes too (kept exact on `canonAdd`; the table is never
64
+ deleted from), because recognition and the join probe it once per span and
65
+ almost every answer is "no such key". It is PERSISTED (`canon_bloom`) in the
66
+ same transaction as the canon rows it covers, stamped with that commit's
67
+ `canon.upto`; an open whose meta disagrees with the stamp (rows written without
68
+ it) rebuilds it by one scan of the h column — seconds on a trained store, paid
69
+ once instead of per process (`test/36`).
47
70
 
48
71
  ## Containment, batching, LRU
49
72
 
@@ -12,6 +12,15 @@ discovering bugs:**
12
12
  leads to accidental bugs, and real-world corpus must not be compromised.
13
13
  - Something that happens due to deduplication, a tie-breaking rule, etc., isn't
14
14
  a bug—and that’s a subtle point.
15
+ - A change in behavior is not a behavioral regression: A change that shifts a
16
+ response from correct to incorrect is not necessarily a regression. Sometimes,
17
+ there may be many other responses that were incorrect but have become correct.
18
+ Overall, there must be a net gain. Budget adjustments or computational
19
+ optimizations often require this.
20
+ - Often, a pathological search is not necessarily resolved by a constant
21
+ derivative. A constant derivative is typically a type of short-circuit
22
+ breaker; it can cause truncation and is not necessarily a budget-related
23
+ measure. Therefore, it must be used wisely.
15
24
 
16
25
  ### 1. `score >= threshold` decides identity
17
26
 
@@ -174,8 +183,10 @@ discovering bugs:**
174
183
  the quantities the machine already has (`leadsSomewhere`, the
175
184
  exact-then-canonical identity, `accounted` bytes, the ladder, `hubBound`),
176
185
  from which the reach of a gap, the offer of a hop, the depth of a join and the
177
- scope of a substitution are CONSEQUENCES, not four separate decisions. Nothing
178
- in this repository is that today.
186
+ scope of a substitution are CONSEQUENCES, not four separate decisions. Where
187
+ the repository stands against it — the engine (`closeOver`), which of the four
188
+ follow from the law, and the one that does not, with the reason — is stated in
189
+ `docs/architecture/closure.md`, and nowhere else.
179
190
  - **THE STANDARD A CHANGE MUST MEET:** state which consequence it is, and show
180
191
  it following from the law. A change that cannot be stated that way is not
181
192
  ready.
@@ -13,22 +13,24 @@ Guards honest silence, determinism, and every pinned contract. Silence:
13
13
  unrelated queries ground to nothing (`test/28`, `50`, `56`, `67`, `76`, `84`).
14
14
  Determinism: same seed + deposit order + query gives byte-identical answer
15
15
  (`test/20`). Every invariant is pinned, the closure law included
16
- (`test/133`–`147`). §14–25 (pipeline), §64 (derived thresholds), AGENTS.md §2
16
+ (`test/133`–`151`). §14–25 (pipeline), §64 (derived thresholds), AGENTS.md §2
17
17
  invariants 1–5.
18
18
 
19
19
  ## 2 — Work accounting (profiler)
20
20
 
21
- ```bash
22
- node bench/profile-inference.mjs # add [n] to limit probes
23
- node bench/profile-inference.mjs --trace # trace is a debugging aid, not product
21
+ ```js
22
+ const mind = new Mind({ profile: true }); // meter attached per response
23
+ await mind.respondText(q); // then read mind.lastCost
24
+ console.log(formatReport(mind.lastCost)); // sumReports() over several
24
25
  ```
25
26
 
26
- Guards without trace: counters exact and diffable between COLD runs; phases nest
27
- (not disjoint — each phase is charged by its own layer); shared analyses charged
28
- to themselves, not to the first toucher; millisecond fields are
29
- non-deterministic hints only. With `--trace`, recognition idempotence still
30
- holds (`test/42`). `src/meter.ts`, `docs/architecture/meter.md`, §55,
31
- `AGENTS.md` §6.
27
+ The public path is the harness (`AGENTS.md` §6); there is no separate bench
28
+ script. Guards without trace: counters exact and diffable between COLD runs;
29
+ phases nest (not disjoint — each phase is charged by its own layer); shared
30
+ analyses charged to themselves, not to the first toucher; millisecond fields are
31
+ non-deterministic hints only. With an `inspectRationale` callback attached,
32
+ recognition idempotence still holds (`test/42`). `src/meter.ts`,
33
+ `docs/architecture/meter.md`, §55, `AGENTS.md` §6.
32
34
 
33
35
  ## 3 — Dependency footprint
34
36
 
@@ -15,8 +15,15 @@ consumes them directly; any site whose bytes overlap a computed span is masked
15
15
  - `formRules` follow continuation edges (`GraphSearch.formRules`): each hop
16
16
  costs `STEP` (1). Forks across all continuations up to the hub bound;
17
17
  disambiguation is distributional, not heuristic.
18
- - Edge-less forms may hop via a halo sibling (`conceptHop` / `resolveConcepts`)
19
- at `CONCEPT` (10), borrowing a synonym's continuation.
18
+ - Edge-less forms may hop via a halo sibling (`conceptHop` / `offerConcepts`) at
19
+ `CONCEPT` (10), borrowing a synonym's continuation.
20
+ - A span's cheapest completion DOMINATES the rest (`buildSearch`): a form or
21
+ completion of `[i, j)` whose cost has reached that of a completion of `[i, j)`
22
+ already yielded fires no rule (`searchDominated`). Coverage is positional, so
23
+ only the cheapest matters to the goal; the byte rules (fuse, splice, join)
24
+ fire from the completion the search would stand on, never from every
25
+ alternative it reached. It also makes the first hop's stop-here
26
+ (`STEP + CONCEPT`) a real horizon for the chain.
20
27
 
21
28
  ## Gate — `leadsSomewhere` (`src/mind/traverse.ts`)
22
29
 
@@ -36,11 +43,18 @@ and are filtered during recognition.
36
43
  The cover reports `moves` (its derivation's discrete work) and `accounted`; the
37
44
  ladder prices both.
38
45
 
39
- ## Pre-resolution (`src/mind/mechanisms/cover.ts`)
46
+ ## Licensed premises (`src/mind/mechanisms/cover.ts`, `Licence` in `graph-search.ts`)
40
47
 
41
- `resolveConcepts` and `resolveConnectors` pre-resolve the async maps the
42
- synchronous search cannot gather: concept targets and learnt connectors, keyed
43
- by node pair. Bridges (`bridge`) splice connectors between rewrites.
48
+ The synchronous search cannot run the async reads two of its rules need — a
49
+ concept target (a halo lookup) and a learnt connector between two answers (a
50
+ `bridge`). `offerConcepts` and `offerConnectors` OFFER the keys up front (cheap:
51
+ `hasNext`, the touching-site pairs and the N-ary allowances); the search ASKS
52
+ for an offered key only where it reaches it — a connector when its splice's two
53
+ premises meet, a concept target when the hop's asking form (held at the hop's
54
+ own cost) is popped. A cover that asked is provisional: `cover.run` grants the
55
+ asked keys and covers again, until a cover asks nothing — which is then the
56
+ cover every key resolved in advance would have made. The joins licensed by
57
+ ask-free rounds are kept across the re-covers.
44
58
 
45
59
  ## Provenance
46
60
 
@@ -54,3 +68,6 @@ independent evidence streams meet at one anchor — see
54
68
 
55
69
  - `test/09-edges.test.mjs` — edge following and hop semantics
56
70
  - `test/19-nd.test.mjs` — form rules and multi-hop chains
71
+ - `test/151` — connectors and concept hops resolved where the search reaches
72
+ them; a span's cheapest completion dominates (a hub's degree generates no
73
+ work)
package/jsr.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "$schema": "https://jsr.io/schema/config-file.v1.json",
3
3
  "name": "@hviana/sema",
4
- "version": "0.9.0",
4
+ "version": "0.9.2",
5
5
  "exports": "./src/index.ts"
6
6
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@hviana/sema",
3
- "version": "0.9.0",
3
+ "version": "0.9.2",
4
4
  "description": "Sema: a non-parametric, instance-based reasoning system.",
5
5
  "repository": {
6
6
  "type": "git",
package/src/alu/README.md CHANGED
@@ -238,8 +238,16 @@ const sync = await prefetchResonance(resonance, spans);
238
238
  ```
239
239
 
240
240
  The synchronous op callbacks never await — they read from the pre-resolved
241
- snapshots. The public `Mind.compute(name, operands)` path pre-resolves every
242
- symbol span before the synchronous kernel runs, using the same discipline.
241
+ snapshots. A named operation applied to the query's operand stream resolves
242
+ opposites ON DEMAND (`withOppositesOnDemand`). The kernel runs against a
243
+ snapshot of the opposites resolved so far. When it asked for one of its symbol
244
+ operands that is not resolved yet, that one is resolved through the host and the
245
+ pure kernel runs again. The result is the eager prefetch's, but only the
246
+ polymorphic inverse reads opposites. Before this, every symbol operand of ANY
247
+ operation paid one host call, and on Sema each is a halo-index query (30–200 ms
248
+ of a plain dialogue turn's parse that computed nothing). The public
249
+ `Mind.compute(name, operands)` path pre-resolves every symbol span before the
250
+ synchronous kernel runs, using the same discipline.
243
251
 
244
252
  ### The mind loop
245
253
 
@@ -51,6 +51,7 @@ export {
51
51
  prefetchOpposites,
52
52
  prefetchRecognisedOps,
53
53
  prefetchResonance,
54
+ withOppositesOnDemand,
54
55
  } from "./resonance.js";
55
56
 
56
57
  export {
@@ -45,10 +45,9 @@ import type { Alu } from "./alu.js";
45
45
  import {
46
46
  type AluResonance,
47
47
  type ConceptAnchor,
48
- prefetchOpposites,
49
48
  prefetchResonance,
49
+ withOppositesOnDemand,
50
50
  } from "./resonance.js";
51
- import { NO_RESONANCE } from "./operation.js";
52
51
  import { int, real, symbol, symbolSpans, type Value } from "./value.js";
53
52
  import { nonSpaceRuns } from "./text.js";
54
53
  import { bytesEqual, latin1 } from "../../bytes.js";
@@ -573,10 +572,11 @@ export class QueryParser {
573
572
  const symbols = picked.flatMap((t, k) =>
574
573
  t.kind === "term" ? [args[k]] : []
575
574
  );
576
- const resonance = symbols.length > 0
577
- ? await prefetchOpposites(this.resonance, symbols)
578
- : NO_RESONANCE;
579
- const bytes = this.alu.applyBytes(name, args, resonance);
575
+ const bytes = await withOppositesOnDemand(
576
+ this.resonance,
577
+ symbols,
578
+ (resonance) => this.alu.applyBytes(name, args, resonance),
579
+ );
580
580
  if (bytes === null) return null;
581
581
  if (
582
582
  symbols.length === args.length && args.some((a) => bytesEqual(bytes, a))
@@ -104,6 +104,48 @@ export async function prefetchOpposites(
104
104
  };
105
105
  }
106
106
 
107
+ /** {@link prefetchOpposites} resolved ON DEMAND: run `apply` against a
108
+ * synchronous snapshot that answers only the opposites already resolved, and
109
+ * when the computation ASKED for one of `symbols` that is not yet resolved,
110
+ * resolve it through the host and run `apply` again — until a run asks for
111
+ * nothing new. The result is `apply(prefetchOpposites(resonance, symbols))`
112
+ * exactly: an opposite outside `symbols` reads null in both, every one inside
113
+ * that the computation reads is the host's answer in both, and `apply` is
114
+ * pure, so a run that read the same answers returns the same bytes. What it
115
+ * saves is every host call no computation reads: only the polymorphic
116
+ * inverse reads opposites, yet the eager prefetch paid one per symbol operand
117
+ * for ANY operation — on SEMA a halo-index query each, measured at 30-200 ms
118
+ * of a plain dialogue turn's parse that computed nothing. */
119
+ export async function withOppositesOnDemand<R>(
120
+ resonance: AluResonance,
121
+ symbols: Iterable<Uint8Array>,
122
+ apply: (sync: ResonanceSync) => R,
123
+ ): Promise<R> {
124
+ const allowed = new Set<string>();
125
+ for (const bytes of symbols) allowed.add(latin1(bytes));
126
+ const table = new Map<string, Uint8Array | null>();
127
+ const asked = new Map<string, Uint8Array>();
128
+ const sync: ResonanceSync = {
129
+ opposite: (bytes: Uint8Array) => {
130
+ const key = latin1(bytes);
131
+ if (!allowed.has(key)) return null;
132
+ const known = table.get(key);
133
+ if (known !== undefined) return known;
134
+ asked.set(key, bytes);
135
+ return null;
136
+ },
137
+ recogniseOp: () => null,
138
+ };
139
+ for (;;) {
140
+ const out = apply(sync);
141
+ if (asked.size === 0) return out;
142
+ for (const [key, bytes] of asked) {
143
+ table.set(key, (await resonance.opposite(bytes)) ?? null);
144
+ }
145
+ asked.clear();
146
+ }
147
+ }
148
+
107
149
  /** Pre-resolve BOTH capabilities a computation may need synchronously — the
108
150
  * resonant opposite of a symbol (for the polymorphic inverse) AND the operation
109
151
  * a symbol's MEANING names (for a higher-order nd op's function argument) — over