@hviana/sema 0.8.0 → 0.8.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (119) hide show
  1. package/AGENTS.md +22 -1
  2. package/DATASETS.md +1 -1
  3. package/dist/example/train_base/config.js +2 -2
  4. package/dist/example/train_base/corpora/massive.js +1 -1
  5. package/dist/example/train_base/readers.js +1 -1
  6. package/dist/src/config.d.ts +17 -0
  7. package/dist/src/config.js +18 -0
  8. package/dist/src/geometry.d.ts +10 -10
  9. package/dist/src/geometry.js +25 -24
  10. package/dist/src/meter.d.ts +29 -12
  11. package/dist/src/meter.js +58 -14
  12. package/dist/src/mind/attention.js +12 -12
  13. package/dist/src/mind/bridge.d.ts +8 -8
  14. package/dist/src/mind/bridge.js +33 -32
  15. package/dist/src/mind/corpus.d.ts +40 -0
  16. package/dist/src/mind/corpus.js +149 -0
  17. package/dist/src/mind/graph-search.d.ts +7 -8
  18. package/dist/src/mind/graph-search.js +244 -32
  19. package/dist/src/mind/index.d.ts +3 -1
  20. package/dist/src/mind/index.js +1 -0
  21. package/dist/src/mind/junction.d.ts +1 -1
  22. package/dist/src/mind/junction.js +8 -8
  23. package/dist/src/mind/learning.js +36 -35
  24. package/dist/src/mind/match.d.ts +8 -3
  25. package/dist/src/mind/match.js +156 -71
  26. package/dist/src/mind/mechanisms/cast.js +18 -2
  27. package/dist/src/mind/mechanisms/cover.js +19 -12
  28. package/dist/src/mind/mechanisms/prefix-completion.js +24 -24
  29. package/dist/src/mind/mechanisms/recall.js +38 -40
  30. package/dist/src/mind/mechanisms/reference.js +16 -16
  31. package/dist/src/mind/mind.d.ts +61 -7
  32. package/dist/src/mind/mind.js +72 -2
  33. package/dist/src/mind/pipeline-mechanism.d.ts +10 -8
  34. package/dist/src/mind/pipeline-mechanism.js +25 -21
  35. package/dist/src/mind/pipeline.d.ts +9 -9
  36. package/dist/src/mind/pipeline.js +49 -29
  37. package/dist/src/mind/primitives.d.ts +5 -5
  38. package/dist/src/mind/primitives.js +5 -5
  39. package/dist/src/mind/reasoning.d.ts +5 -1
  40. package/dist/src/mind/reasoning.js +54 -1
  41. package/dist/src/mind/recognition.d.ts +14 -13
  42. package/dist/src/mind/recognition.js +23 -23
  43. package/dist/src/mind/resonance.js +21 -21
  44. package/dist/src/mind/traverse.d.ts +54 -52
  45. package/dist/src/mind/traverse.js +83 -73
  46. package/dist/src/mind/types.d.ts +26 -4
  47. package/dist/src/store.d.ts +12 -12
  48. package/dist/src/store.js +12 -12
  49. package/docs/INDEX.md +2 -2
  50. package/docs/architecture/exact-vs-approximate.md +2 -1
  51. package/docs/architecture/fold-contract.md +1 -1
  52. package/docs/failures/tempting-but-wrong.md +33 -5
  53. package/docs/harness/gates.md +7 -7
  54. package/example/train_base/config.ts +2 -2
  55. package/example/train_base/corpora/massive.ts +1 -1
  56. package/example/train_base/readers.ts +1 -1
  57. package/jsr.json +1 -1
  58. package/package.json +1 -1
  59. package/src/config.ts +35 -0
  60. package/src/geometry.ts +25 -24
  61. package/src/meter.ts +61 -14
  62. package/src/mind/attention.ts +12 -12
  63. package/src/mind/bridge.ts +33 -32
  64. package/src/mind/corpus.ts +202 -0
  65. package/src/mind/graph-search.ts +261 -31
  66. package/src/mind/index.ts +8 -1
  67. package/src/mind/junction.ts +8 -8
  68. package/src/mind/learning.ts +36 -35
  69. package/src/mind/match.ts +163 -73
  70. package/src/mind/mechanisms/cast.ts +17 -1
  71. package/src/mind/mechanisms/cover.ts +18 -12
  72. package/src/mind/mechanisms/prefix-completion.ts +24 -24
  73. package/src/mind/mechanisms/recall.ts +38 -40
  74. package/src/mind/mechanisms/reference.ts +16 -16
  75. package/src/mind/mind.ts +129 -7
  76. package/src/mind/pipeline-mechanism.ts +25 -21
  77. package/src/mind/pipeline.ts +63 -38
  78. package/src/mind/primitives.ts +5 -5
  79. package/src/mind/reasoning.ts +55 -0
  80. package/src/mind/recognition.ts +23 -23
  81. package/src/mind/resonance.ts +21 -21
  82. package/src/mind/traverse.ts +83 -73
  83. package/src/mind/types.ts +30 -4
  84. package/src/store.ts +20 -20
  85. package/test/08-storage.test.mjs +1 -1
  86. package/test/100-complete-grounding-trace.test.mjs +109 -0
  87. package/test/101-alignment-gap-bound.test.mjs +106 -0
  88. package/test/102-production-composes-at-scale.test.mjs +110 -0
  89. package/test/103-alignment-gap-budget.test.mjs +89 -0
  90. package/test/104-composition-is-reported.test.mjs +90 -0
  91. package/test/105-derive-through-reports-its-refusal.test.mjs +113 -0
  92. package/test/106-the-join-fires.test.mjs +94 -0
  93. package/test/107-the-join-is-counted.test.mjs +81 -0
  94. package/test/108-the-join-chains.test.mjs +78 -0
  95. package/test/109-the-pivot-is-counted.test.mjs +60 -0
  96. package/test/110-the-reasoner-stops-when-the-question-is-answered.test.mjs +91 -0
  97. package/test/111-the-cover-assembly-is-counted.test.mjs +74 -0
  98. package/test/112-the-exploration-does-not-grow-with-the-hub.test.mjs +89 -0
  99. package/test/113-the-rationale-payload-is-bounded.test.mjs +84 -0
  100. package/test/114-alignment-budget-is-per-sweep.test.mjs +93 -0
  101. package/test/116-the-extension-is-gated-by-the-pipelines-own-remainder.test.mjs +100 -0
  102. package/test/117-corpus-search.test.mjs +171 -0
  103. package/test/14-scaling.test.mjs +10 -7
  104. package/test/35-prefix-edge.test.mjs +1 -1
  105. package/test/40-choosenext-scale-guard.test.mjs +16 -17
  106. package/test/56-bridge-identity-admission.test.mjs +6 -6
  107. package/test/70-prefix-completion.test.mjs +4 -3
  108. package/test/72-prefix-candidate-supply.test.mjs +3 -3
  109. package/test/73-scaffolding-only-bridge-abstains.test.mjs +6 -6
  110. package/test/75-multiturn-context-optimisation.test.mjs +5 -5
  111. package/test/76-reference-binding.test.mjs +6 -1
  112. package/test/84-composed-answer-honesty.test.mjs +5 -6
  113. package/test/88-dependency-footprint.test.mjs +1 -1
  114. package/test/89-completion-recursion.test.mjs +47 -19
  115. package/test/90-connector-read-cap.test.mjs +10 -8
  116. package/test/93-regime-prediction.test.mjs +10 -10
  117. package/test/94-cross-region-budget.test.mjs +2 -2
  118. package/test/95-wide-resonance-removed.test.mjs +8 -7
  119. package/test/96-bytes-walk-termination.test.mjs +3 -3
package/src/geometry.ts CHANGED
@@ -349,19 +349,20 @@ function bytesToLeaves(
349
349
  * sentences fall would be importing an assumption the architecture rejects.
350
350
  * Random binary must, and does, behave exactly like prose.
351
351
  *
352
- * Every constant is derived (§2.2): the cut mask is W, so a cut is offered once
353
- * per quantum of bytes — which, composed with the minimum below, puts the
354
- * expected segment at minLen + W − 1 ≈ 6 B rather than at W, deliberately (see
355
- * the refutation recorded at `cutRate` in {@link contentLevels}: a segment is
356
- * the flat PHRASE-scale unit the W-ary groups are built from, not a group of W
357
- * children, and forcing E[len] = W costs 15 tests). The minimum is W−1, `canonicalWindows`'s
358
- * straddle neighbour and the write side's own floor for a unit; and the maximum
359
- * is the KEYRING's seat count, because a segment folds as ONE flat node and
360
- * `fold` has exactly that many seats to bind children into. Capping there is
361
- * what keeps the fold light: a segment of 3..seats leaves is a single node,
362
- * where splitting it into W-groups plus a remainder would cost two or three
363
- * and the remainders barely share (measured: partial-arity nodes 504 → 3,590,
364
- * and total distinct nodes 8,142 → 9,712, when segments folded as [W][rest]). */
352
+ * Every constant is derived (thresholds.md): the cut mask is W, so a cut is
353
+ * offered once per quantum of bytes — which, composed with the minimum below,
354
+ * puts the expected segment at minLen + W − 1 ≈ 6 B rather than at W,
355
+ * deliberately (see the refutation recorded at `cutRate` in {@link
356
+ * contentLevels}: a segment is the flat PHRASE-scale unit the W-ary groups are
357
+ * built from, not a group of W children, and forcing E[len] = W costs 15
358
+ * tests). The minimum is W−1, `canonicalWindows`'s straddle neighbour and the
359
+ * write side's own floor for a unit; and the maximum is the KEYRING's seat
360
+ * count, because a segment folds as ONE flat node and `fold` has exactly that
361
+ * many seats to bind children into. Capping there is what keeps the fold light:
362
+ * a segment of 3..seats leaves is a single node, where splitting it into
363
+ * W-groups plus a remainder would cost two or three and the remainders barely
364
+ * share (measured: partial-arity nodes 504 → 3,590, and total distinct nodes
365
+ * 8,142 → 9,712, when segments folded as [W][rest]). */
365
366
  /** {@link contentBoundaries} plus, for each cut, its LEVEL — how deep in the
366
367
  * tree that cut reaches.
367
368
  *
@@ -622,16 +623,16 @@ export function knownPrefixLength(
622
623
  * correct boundary. Pass them through from `perceive`; the geometry
623
624
  * computes the stable prefix internally.
624
625
  *
625
- * `boundaries` is the CALLER-computed stable-prefix boundary set (§10.3):
626
- * strictly-increasing proper byte offsets, each the length of a prefix that
627
- * is already a stored whole-stream form. When given, the fold splits into
628
- * the segments between consecutive boundaries — each folded independently,
629
- * exactly as it folded when it was learned — and the segment roots join
630
- * LEFT-NESTED (((s₀·s₁)·s₂)…), so every learnt cumulative-context root
631
- * reappears as an identical subtree (and, by hash-consing, the very same
632
- * node) inside the grown stream. This is what lets a conversation's next
633
- * turn extend perception instead of refolding it: identical prefixes
634
- * produce identical subtrees regardless of what follows them. */
626
+ * `boundaries` is the CALLER-computed stable-prefix boundary set
627
+ * (fold-contract.md): strictly-increasing proper byte offsets, each the length
628
+ * of a prefix that is already a stored whole-stream form. When given, the fold
629
+ * splits into the segments between consecutive boundaries — each folded
630
+ * independently, exactly as it folded when it was learned — and the segment
631
+ * roots join LEFT-NESTED (((s₀·s₁)·s₂)…), so every learnt cumulative-context
632
+ * root reappears as an identical subtree (and, by hash-consing, the very same
633
+ * node) inside the grown stream. This is what lets a conversation's next turn
634
+ * extend perception instead of refolding it: identical prefixes produce
635
+ * identical subtrees regardless of what follows them. */
635
636
  export function bytesToTree(
636
637
  space: Space,
637
638
  alphabet: Alphabet,
@@ -962,7 +963,7 @@ function flatFold(
962
963
  return { tree: sema(gist, null, kids), len: n };
963
964
  }
964
965
 
965
- /** The stable-prefix segmented fold (§10.3). Each segment between
966
+ /* * The stable-prefix segmented fold (fold-contract.md). Each segment between
966
967
  * consecutive boundaries folds PLAINLY and independently; segment roots
967
968
  * join left-nested, and only the final root is normalized (the linear-fold
968
969
  * contract: one normalize per perception). A segment's own inner splits
package/src/meter.ts CHANGED
@@ -8,8 +8,8 @@
8
8
  // Four contracts, all load-bearing:
9
9
  //
10
10
  // 1. NEVER READ BY INFERENCE. No counter may reach a decision, a threshold,
11
- // or an ordering. Determinism (AGENTS §2.1) survives only because the
12
- // meter is write-only from the engine's point of view.
11
+ // or an ordering. Determinism (determinism.md) survives
12
+ // only because the meter is write-only from the engine's point of view.
13
13
  // 2. OFF BY DEFAULT, AND FREE WHEN OFF. Every call site is `meter?.x++` on
14
14
  // a null field. Nothing allocates, nothing is keyed, nothing is timed
15
15
  // unless a Meter is attached (`new Mind({ profile: true })`).
@@ -79,9 +79,9 @@ export class Meter {
79
79
  nodeRecords = 0;
80
80
  /** `store.bytes` / `store.bytesPrefix` — one reconstruction request. */
81
81
  byteReads = 0;
82
- /** Bytes actually handed back by those reads — the real I/O volume, and
83
- * the number that exposes an unbounded read (AGENTS §2.8) that a call
84
- * count alone hides. */
82
+ /* * Bytes actually handed back by those reads — the real I/O volume, and the
83
+ * number that exposes an unbounded read (bounded-reads.md) that a call count
84
+ * alone hides. */
85
85
  bytesRead = 0;
86
86
  /** `store.contentLen`. */
87
87
  lenReads = 0;
@@ -178,10 +178,10 @@ export class Meter {
178
178
  junctionPops = 0;
179
179
  /** Ascents that ended by EXHAUSTING the expansion budget rather than by
180
180
  * deciding — the walk abstained and the caller silently fell through to a
181
- * lower tier of the ladder (§2.13: a degradation nothing else reports).
182
- * It rises the moment a SHARED budget is drained by an earlier walk, which
183
- * is what makes "this tier answered nothing" distinguishable from "this
184
- * tier never got to look". */
181
+ * lower tier of the ladder — honest degradation, and nothing else reports it
182
+ * (INVARIANTS.md). It rises the moment a SHARED budget is drained by an
183
+ * earlier walk, which is what makes "this tier answered nothing"
184
+ * distinguishable from "this tier never got to look". */
185
185
  junctionBudgetExhausted = 0;
186
186
  /** Arbitrary byte spans whose distributional company was VSA-bundled from
187
187
  * existing episode halos. */
@@ -206,6 +206,53 @@ export class Meter {
206
206
  /** Candidates the decider weighed. */
207
207
  candidates = 0;
208
208
 
209
+ // ── Graph search: the fact join (DIRECTION) ─────────────────────────────
210
+ //
211
+ // The join's outcome was observable ONLY through the rationale, and the
212
+ // rationale PERTURBS the search (measured: appending text to a refusal note
213
+ // changed a traced answer). These four counters are the untraced view — the
214
+ // same surface every other work counter uses, incremented where the decision
215
+ // is made, never behind a trace guard.
216
+ /** `deriveThrough` yielded — a fact was reached through the subject the query
217
+ * never named. */
218
+ joinFired = 0;
219
+ /** Refused: no key names the entity and the tail together. (A key that
220
+ * resolves but leads nowhere is not "refused" — it is not the relation, so
221
+ * the scan simply moves on; there is no counter for a case the loop cannot
222
+ * reach.) */
223
+ joinNoKey = 0;
224
+ /** Refused: the fact contains no entity that leads anywhere. */
225
+ joinNoEntity = 0;
226
+
227
+ // ── Mind: the multi-hop pivot (EXTENSION) ───────────────────────────────
228
+ //
229
+ // `pivotStep` was observable only through the rationale, and the rationale
230
+ // perturbs the search (measured). How far the reasoner hopped is a
231
+ // BEHAVIOUR, so it needs an untraced view: one counter, incremented where the
232
+ // step is emitted.
233
+ /** Times the reasoner pivoted on a span its answer contains and stepped
234
+ * across that fact. */
235
+ pivotSteps = 0;
236
+
237
+ // ── Mind: the cover's connector assembly (LIMIT) ────────────────────────
238
+ //
239
+ // The cover's `run` is 91% of a hub query's time (`"Hello."`: 2.7 s of 3.0 s)
240
+ // and holds its ~270 MB peak, and none of it was countable: `searchPushes`
241
+ // and `candidates` do not see the connector assembly. These two counters are
242
+ // the untraced view of it.
243
+ /** `bridge` calls the cover makes assembling connectors (pairwise + n-ary). */
244
+ coverBridges = 0;
245
+ /** Continuations a CHAIN hop offered the search. Bounded by the question
246
+ * (`ceil(queryLen / W)`) rather than by the corpus's fan-out — measured on a
247
+ * hub of degree 1083, offering every continuation grew the chart to 3113 outs
248
+ * and cost a 270 MB peak / 256 MB OOM for a two-word question. */
249
+ chainOffers = 0;
250
+ /** Σ byte-allowance the n-ary interior passes those bridges. The allowance
251
+ * is `middleBytes + (m + 1) * W` — every intermediate answer's bytes plus
252
+ * one window of glue per joint — so it is the quantity that grows with a hub
253
+ * query's answers, and the first thing to read when the peak moves. */
254
+ coverAllowanceBytes = 0;
255
+
209
256
  // ── Phases ──────────────────────────────────────────────────────────────
210
257
 
211
258
  private readonly _phases = new Map<string, PhaseCost>();
@@ -238,11 +285,11 @@ export class Meter {
238
285
  }
239
286
  }
240
287
 
241
- /** Time one SYNCHRONOUS phase. The sync/async seam (§2.10) is a real
242
- * contract — perception, recognition and the graph search are synchronous —
243
- * so a synchronous layer must not be wrapped in `time`'s promise just to be
244
- * measured: that would make the profiled path await where the unprofiled
245
- * one does not, and a meter never changes what a layer computes. */
288
+ /* * Time one SYNCHRONOUS phase. The sync/async seam is a real contract
289
+ * (meter.md) — perception, recognition and the graph search are synchronous —
290
+ * so a synchronous layer must not be wrapped in `time`'s promise just to be
291
+ * measured: that would make the profiled path await where the unprofiled one
292
+ * does not, and a meter never changes what a layer computes. */
246
293
  timeSync<T>(phase: string, fn: () => T): T {
247
294
  const before = this.snapshot();
248
295
  const t = performance.now();
@@ -1941,10 +1941,10 @@ export function canonicalChunkId(
1941
1941
  // CAST lost a point of attention it needed (test/29 D1/D2).
1942
1942
  //
1943
1943
  // So scan every offset and prefer an anchor that still discriminates: not
1944
- // saturated, and among those the one reaching the FEWEST contexts (§2.7,
1945
- // corpus-global). Only when every window in the region saturates does the
1946
- // old generalising choice stand — there is then no discriminative anchor to
1947
- // find, and abstaining is the honest outcome.
1944
+ // saturated, and among those the one reaching the FEWEST contexts
1945
+ // (commonality.md, corpus-global). Only when every window in the region
1946
+ // saturates does the old generalising choice stand — there is then no
1947
+ // discriminative anchor to find, and abstaining is the honest outcome.
1948
1948
  let discId: number | null = null;
1949
1949
  let discReached = Infinity;
1950
1950
  let fallback: number | null = null;
@@ -2649,16 +2649,16 @@ async function crossRegionVotes(
2649
2649
  const consumed = new Set<number>();
2650
2650
  let probes = 0;
2651
2651
  // When atoms themselves are hubs (atomIsHub — a single byte reaches ≥ √N
2652
- // contexts, §2.8's own predicate), the corpus is large enough that the
2653
- // cross-region junction walks are dominated by the drift through common
2654
- // content's ancestry. Each of k candidate pairs otherwise spends its own
2655
- // √N·W budget (profiled: 160,210 junction pops, 31% of think at
2656
- // N = 325,608), and a cumulative dialogue multiplies bounded work into tens
2657
- // of seconds. The structural walk is therefore given ONE k·W allowance per
2652
+ // contexts, bounded-reads.md's own predicate), the corpus is large enough
2653
+ // that the cross-region junction walks are dominated by the drift through
2654
+ // common content's ancestry. Each of k candidate pairs otherwise spends its
2655
+ // own √N·W budget (profiled: 160,210 junction pops, 31% of think at N =
2656
+ // 325,608), and a cumulative dialogue multiplies bounded work into tens of
2657
+ // seconds. The structural walk is therefore given ONE k·W allowance per
2658
2658
  // evidence tier, shared across every pair — k pairs × W phrase-scale levels,
2659
2659
  // the minimal exact check; a pair whose container is not reached within it
2660
- // falls through to the resonance tier (the ANN proposes what the shallow
2661
- // walk no longer exhaustively scans, §2.3).
2660
+ // falls through to the resonance tier (the ANN proposes what the shallow walk
2661
+ // no longer exhaustively scans, exact-vs-approximate.md).
2662
2662
  //
2663
2663
  // Below atomIsHub the store is small and atoms still discriminate, so the
2664
2664
  // walks keep exhaustive exact traversal (per-walk √N·W) — the shared budget
@@ -148,23 +148,23 @@ export function dismissedKnownContent(
148
148
  return false;
149
149
  }
150
150
 
151
- // The seeded aligner this file used to own now lives in the shared match
152
- // family as {@link alignAround} — the frame reading (match.ts) reads the same
153
- // gaps and asks the OPPOSITE question of them (see AlignGap's own doc). Two
154
- // consumers, one definition (AGENTS §2.5); the bridge's reading is unchanged.
151
+ // The seeded aligner this file used to own now lives in the shared match family
152
+ // as {@link alignAround} — the frame reading (match.ts) reads the same gaps and
153
+ // asks the OPPOSITE question of them (see AlignGap's own doc). Two consumers,
154
+ // one definition (factored-machinery.md); the bridge's reading is unchanged.
155
155
  const align = alignAround;
156
156
 
157
157
  /** Recall's corroborated-substitution bridge — see the module comment.
158
158
  * Returns the best bridged grounding proposal, or null. */
159
159
  /** `proposed` is a THUNK, not a list: the bridge's own cheap gates (the
160
- * two-quantum query floor and the O(|query|) stored-window anchor scan)
161
- * decide whether ANY candidate can be aligned, and they need no proposals
162
- * to do it. Resolving the caller's proposals eagerly meant recall paid its
163
- * exhaustive whole-index resonance — the most expensive single act on the
164
- * refusal path — for every query, including the ones whose windows the
165
- * store has never seen and which the anchor scan rejects outright. Same
166
- * investment discipline the mechanism floors follow (AGENTS §2.6): never
167
- * compute a shared analysis just to discard it. */
160
+ * two-quantum query floor and the O(|query|) stored-window anchor scan) decide
161
+ * whether ANY candidate can be aligned, and they need no proposals to do it.
162
+ * Resolving the caller's proposals eagerly meant recall paid its exhaustive
163
+ * whole-index resonance — the most expensive single act on the refusal path —
164
+ * for every query, including the ones whose windows the store has never seen
165
+ * and which the anchor scan rejects outright. Same investment discipline the
166
+ * mechanism floors follow (mechanism-market.md): never compute a shared
167
+ * analysis just to discard it. */
168
168
  export async function substitutionBridge(
169
169
  ctx: MindContext,
170
170
  query: Uint8Array,
@@ -290,12 +290,12 @@ async function bridgeImpl(
290
290
  );
291
291
  return null;
292
292
  }
293
- // NO DISCRIMINATING LITERAL EVIDENCE — abstain (§2.13). A bridge grounds
294
- // through the literal spans it did NOT substitute; those anchors are the
295
- // whole of its evidence. When every one of them is SATURATED — containment
296
- // clamped at the √N hub bound, i.e. the window is corpus-global scaffolding
297
- // — the query's unsubstituted part discriminates nothing, and the single
298
- // substituted span is carrying the entire semantic load. That is not a
293
+ // NO DISCRIMINATING LITERAL EVIDENCE — abstain (INVARIANTS.md). A bridge
294
+ // grounds through the literal spans it did NOT substitute; those anchors are
295
+ // the whole of its evidence. When every one of them is SATURATED —
296
+ // containment clamped at the √N hub bound, i.e. the window is corpus-global
297
+ // scaffolding — the query's unsubstituted part discriminates nothing, and the
298
+ // single substituted span is carrying the entire semantic load. That is not a
299
299
  // corroborated bridge; it is a template match, and it FABRICATES.
300
300
  //
301
301
  // Measured on the trained store (hubBound 571). "What is the capital of"
@@ -310,7 +310,8 @@ async function bridgeImpl(
310
310
  // them silent and cannot be credited for them.
311
311
  //
312
312
  // This introduces NO new threshold: `bound` is the same √N reading of "hub"
313
- // the anchor scan already clamps its own containment read to (§2.2, §2.7).
313
+ // the anchor scan already clamps its own containment read to (thresholds.md,
314
+ // commonality.md).
314
315
  if (allWindowsAreScaffolding(ctx, query)) {
315
316
  ctx.trace?.step(
316
317
  "substitutionBridge",
@@ -353,8 +354,8 @@ async function bridgeImpl(
353
354
  //
354
355
  // The question every gap poses is "may the two forms differ HERE without
355
356
  // differing in what they SAY?", and that is the discriminative-vs-
356
- // scaffolding question AGENTS §2.7 names, over the CORPUS-GLOBAL
357
- // population. It already has one definition — `dominates(reachOf(...), N)`,
357
+ // scaffolding question commonality.md names, over the CORPUS-GLOBAL
358
+ // population. It already has one definition — `dominates(reachOf(...), N)`,
358
359
  // the same gate confluence's filler test uses ("scaffolding never binds").
359
360
  // Nothing new is derived here; the bar is read, not invented.
360
361
  //
@@ -366,17 +367,17 @@ async function bridgeImpl(
366
367
  // climb's own definition of non-discriminative), or it resolves to a
367
368
  // majority of the corpus's contexts. "the process of ", " is the ".
368
369
  //
369
- // THE READING MATTERS, not just the population (AGENTS §2.7). This
370
- // deliberately does NOT go through `reachOf`, which maps BOTH "saturated"
371
- // and "reaches nothing" to Infinity. For IDF weighting those are the same
372
- // thing (no usable identity evidence); for THIS question they are
373
- // opposites — a window reaching nothing is novel content, the most
374
- // discriminative material there is, and reading it as Infinity would call
375
- // it scaffolding. Measured: with `reachOf`, "Is water wet?" was answered
376
- // with "No, heavy water is not wet." — "heav"/"eavy" occur once, reach no
377
- // edge-bearing ancestor, and were written off as filler. So an
378
- // empty-rooted window is NEVER explained, and neither is an untrained one
379
- // (the same principle attestedQ applies to the query side).
370
+ // THE READING MATTERS, not just the population — see commonality.md. This
371
+ // deliberately does NOT go through `reachOf`, which maps BOTH "saturated" and
372
+ // "reaches nothing" to Infinity. For IDF weighting those are the same thing
373
+ // (no usable identity evidence); for THIS question they are opposites — a
374
+ // window reaching nothing is novel content, the most discriminative material
375
+ // there is, and reading it as Infinity would call it scaffolding. Measured:
376
+ // with `reachOf`, "Is water wet?" was answered with "No, heavy water is not
377
+ // wet." — "heav"/"eavy" occur once, reach no edge-bearing ancestor, and were
378
+ // written off as filler. So an empty-rooted window is NEVER explained, and
379
+ // neither is an untrained one (the same principle attestedQ applies to the
380
+ // query side).
380
381
  const reachMemo = sharedReachMemo(ctx);
381
382
  const explainedSpan = (
382
383
  bytes: Uint8Array,
@@ -0,0 +1,202 @@
1
+ // corpus.ts — read the trained memory back out of the DAG, AS DATA.
2
+ //
3
+ // A trained experience pair IS one continuation edge: `src` is the context that
4
+ // was deposited, `dst` is what the mind learnt follows it. Reading them back
5
+ // uses the store's own structure and its own indexes — no auxiliary index is
6
+ // built, nothing is written, and NOTHING HERE KNOWS ABOUT TEXT: this layer takes
7
+ // bytes and returns bytes. The text case is one helper on the Mind
8
+ // (`searchCorpusText`), which encodes, calls this, and decodes.
9
+ //
10
+ // WHERE EACH STAGE COMES FROM (ported from the demo's `explore.ts`, which
11
+ // hand-rolled its own resolution):
12
+ //
13
+ // 1. PERCEIVE, CONTENT-ADDRESS and ADMIT the query — `recognise()`, the SAME
14
+ // machinery an answer goes through. It returns the sites: the query spans
15
+ // that content-addressed to a stored node that can lead somewhere. The
16
+ // demo's own recursive `findLeaf`/`findBranch` walk was a second
17
+ // implementation of exactly this, and it is NOT ported.
18
+ // 2. CLIMB the structural `kid` table from each resolved site to the
19
+ // edge-bearing contexts above it (`edgeAncestors`, traverse.ts), weighting
20
+ // each context by how much query content reached it.
21
+ // 3. READ the continuation off the edge table (`nextFirst`).
22
+ //
23
+ // COST is set by how much of the QUERY resolves, never by the size of the
24
+ // store: the sites are what recognition already found, the climb is bounded by
25
+ // the declared `corpusClimbs`/`corpusContextsPerClimb`, and every store call is
26
+ // a point probe or a capped read. All work is accounted by the store's own
27
+ // meter hooks when a response's meter is open — there is no second instrument.
28
+ //
29
+ // WHAT THIS IS NOT. Exact content addressing, not fuzzy keyword search: a query
30
+ // shares results with a stored note when it shares actual chunk-aligned content
31
+ // with it. An arbitrary mid-word fragment resolves to nothing, and the honest
32
+ // answer there is "nothing matched" — which is why `sampleCorpus` exists, and
33
+ // why the miss is reported as a STATE rather than as prose (the text helper
34
+ // turns it into words).
35
+
36
+ import { recognise } from "./recognition.js";
37
+ import { edgeAncestors } from "./traverse.js";
38
+ import type { MindContext } from "./types.js";
39
+
40
+ /** One stored experience pair, as bytes. */
41
+ export interface CorpusPair {
42
+ context: Uint8Array;
43
+ continuation: Uint8Array;
44
+ contextId: number;
45
+ continuationId: number;
46
+ /** Bytes of the query this pair was matched on — 0 when browsing. */
47
+ matchedBytes: number;
48
+ /** True when the stored bytes ran past the declared preview capacity. */
49
+ contextTruncated: boolean;
50
+ continuationTruncated: boolean;
51
+ }
52
+
53
+ /** Why a search produced no pairs. A STATE, so a caller's own layer can say it
54
+ * in its own words — the byte layer does not speak. */
55
+ export type CorpusMiss = "matched" | "nothing-resolved" | "no-continuations";
56
+
57
+ export interface CorpusResult {
58
+ pairs: CorpusPair[];
59
+ /** Query subtrees that content-addressed to a real stored node. */
60
+ resolved: number;
61
+ /** Distinct edge-bearing contexts the climb reached. */
62
+ reached: number;
63
+ /** Distinct contexts that carry a learnt continuation, store-wide. */
64
+ totalContexts: number;
65
+ /** True when these are browse samples rather than search results. */
66
+ browsed: boolean;
67
+ miss: CorpusMiss;
68
+ }
69
+
70
+ /** A context node as a pair, or null when it carries no continuation. */
71
+ function pairOf(
72
+ ctx: MindContext,
73
+ id: number,
74
+ matchedBytes: number,
75
+ ): CorpusPair | null {
76
+ const outs = ctx.store.nextFirst(id, 1);
77
+ if (outs.length === 0) return null;
78
+ const cap = ctx.cfg.corpusPreviewBytes;
79
+ const context = ctx.store.bytesPrefix(id, cap + 1);
80
+ const continuation = ctx.store.bytesPrefix(outs[0], cap + 1);
81
+ const contextTruncated = context.length > cap;
82
+ const continuationTruncated = continuation.length > cap;
83
+ return {
84
+ context: contextTruncated ? context.subarray(0, cap) : context,
85
+ continuation: continuationTruncated
86
+ ? continuation.subarray(0, cap)
87
+ : continuation,
88
+ contextId: id,
89
+ continuationId: outs[0],
90
+ matchedBytes,
91
+ contextTruncated,
92
+ continuationTruncated,
93
+ };
94
+ }
95
+
96
+ /** Which stored notes does this query reach? BYTES in, BYTES out.
97
+ *
98
+ * Exact content addressing through the machinery that already exists: the
99
+ * query's recognised sites are the resolved subtrees, the climb goes up from
100
+ * the biggest first, and a pair is a context that carries a continuation. */
101
+ export function searchCorpus(
102
+ ctx: MindContext,
103
+ queryBytes: Uint8Array,
104
+ limit?: number,
105
+ ): CorpusResult {
106
+ const store = ctx.store;
107
+ const want = Math.max(
108
+ 1,
109
+ Math.min(limit ?? ctx.cfg.corpusContextsPerClimb, ctx.cfg.corpusLimitMax),
110
+ );
111
+ // A resolved subtree must account for at least one window (W): a single
112
+ // character resolves against almost any store and means nothing. W is the
113
+ // mind's own line between chance and evidence — derived, never declared.
114
+ const floor = ctx.space.maxGroup;
115
+ const resolved = recognise(ctx, queryBytes).sites
116
+ .map((s) => ({ id: s.payload, len: store.contentLen(s.payload, 512) }))
117
+ .filter((r) => r.len >= floor);
118
+ // Biggest first, lowest id breaking ties: a clause is evidence, a character is
119
+ // noise, and equal evidence must not be decided by iteration order.
120
+ const byLength = resolved
121
+ .sort((a, b) => b.len - a.len || a.id - b.id)
122
+ .slice(0, ctx.cfg.corpusClimbs);
123
+
124
+ // Weight each context by how much query content reached it.
125
+ const weight = new Map<number, number>();
126
+ for (const { id, len } of byLength) {
127
+ for (
128
+ const root of edgeAncestors(ctx, id, ctx.cfg.corpusContextsPerClimb).roots
129
+ ) {
130
+ weight.set(root, (weight.get(root) ?? 0) + len);
131
+ }
132
+ }
133
+
134
+ const pairs: CorpusPair[] = [];
135
+ const ranked = [...weight.entries()].sort(
136
+ (a, b) => b[1] - a[1] || a[0] - b[0],
137
+ );
138
+ for (const [id, w] of ranked) {
139
+ if (pairs.length >= want) break;
140
+ const pair = pairOf(ctx, id, w);
141
+ if (pair) pairs.push(pair);
142
+ }
143
+ return {
144
+ pairs,
145
+ resolved: byLength.length,
146
+ reached: weight.size,
147
+ totalContexts: store.edgeSourceCount(),
148
+ browsed: false,
149
+ miss: pairs.length > 0
150
+ ? "matched"
151
+ : weight.size === 0
152
+ ? "nothing-resolved"
153
+ : "no-continuations",
154
+ };
155
+ }
156
+
157
+ /** Browse real pairs, striding the id space so the sample is spread rather than
158
+ * one local cluster. DETERMINISTIC: `from` is the caller's own offset, so
159
+ * browsing twice with different offsets shows different notes without a random
160
+ * draw (the demo drew `Math.random()`, which the engine cannot do — same seed,
161
+ * same order, same query must mean the same answer). */
162
+ export function sampleCorpus(
163
+ ctx: MindContext,
164
+ limit?: number,
165
+ from = 0,
166
+ ): CorpusResult {
167
+ const store = ctx.store;
168
+ const want = Math.max(
169
+ 1,
170
+ Math.min(limit ?? ctx.cfg.corpusContextsPerClimb, ctx.cfg.corpusLimitMax),
171
+ );
172
+ const total = store.nodeCount();
173
+ const probes = ctx.cfg.corpusSampleProbes;
174
+ const floorBytes = ctx.cfg.corpusSampleFloorBytes;
175
+ const pairs: CorpusPair[] = [];
176
+ // EACH CONTEXT AT MOST ONCE. Striding the id space revisits ids when the
177
+ // store is small relative to the probe budget (measured: a 160-node store
178
+ // returned the SAME pair six times for `limit: 6`), and a browse that repeats
179
+ // itself is not a browse. The demo had the same hole; it is invisible only on
180
+ // a store far larger than the probe budget.
181
+ const seen = new Set<number>();
182
+ for (let i = 0; i < probes && pairs.length < want && total > 0; i++) {
183
+ const slot = (i / probes + from) % 1;
184
+ const id = Math.floor(slot * total);
185
+ if (seen.has(id)) continue;
186
+ if (!store.has(id) || !store.hasNext(id)) continue;
187
+ if (store.contentLen(id, floorBytes) < floorBytes) continue;
188
+ const pair = pairOf(ctx, id, 0);
189
+ if (pair) {
190
+ seen.add(id);
191
+ pairs.push(pair);
192
+ }
193
+ }
194
+ return {
195
+ pairs,
196
+ resolved: 0,
197
+ reached: pairs.length,
198
+ totalContexts: store.edgeSourceCount(),
199
+ browsed: true,
200
+ miss: pairs.length > 0 ? "matched" : "no-continuations",
201
+ };
202
+ }