@hviana/sema 0.8.0 → 0.8.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (119) hide show
  1. package/AGENTS.md +22 -1
  2. package/DATASETS.md +1 -1
  3. package/dist/example/train_base/config.js +2 -2
  4. package/dist/example/train_base/corpora/massive.js +1 -1
  5. package/dist/example/train_base/readers.js +1 -1
  6. package/dist/src/config.d.ts +17 -0
  7. package/dist/src/config.js +18 -0
  8. package/dist/src/geometry.d.ts +10 -10
  9. package/dist/src/geometry.js +25 -24
  10. package/dist/src/meter.d.ts +29 -12
  11. package/dist/src/meter.js +58 -14
  12. package/dist/src/mind/attention.js +12 -12
  13. package/dist/src/mind/bridge.d.ts +8 -8
  14. package/dist/src/mind/bridge.js +33 -32
  15. package/dist/src/mind/corpus.d.ts +40 -0
  16. package/dist/src/mind/corpus.js +149 -0
  17. package/dist/src/mind/graph-search.d.ts +7 -8
  18. package/dist/src/mind/graph-search.js +244 -32
  19. package/dist/src/mind/index.d.ts +3 -1
  20. package/dist/src/mind/index.js +1 -0
  21. package/dist/src/mind/junction.d.ts +1 -1
  22. package/dist/src/mind/junction.js +8 -8
  23. package/dist/src/mind/learning.js +36 -35
  24. package/dist/src/mind/match.d.ts +8 -3
  25. package/dist/src/mind/match.js +156 -71
  26. package/dist/src/mind/mechanisms/cast.js +18 -2
  27. package/dist/src/mind/mechanisms/cover.js +19 -12
  28. package/dist/src/mind/mechanisms/prefix-completion.js +24 -24
  29. package/dist/src/mind/mechanisms/recall.js +38 -40
  30. package/dist/src/mind/mechanisms/reference.js +16 -16
  31. package/dist/src/mind/mind.d.ts +61 -7
  32. package/dist/src/mind/mind.js +72 -2
  33. package/dist/src/mind/pipeline-mechanism.d.ts +10 -8
  34. package/dist/src/mind/pipeline-mechanism.js +25 -21
  35. package/dist/src/mind/pipeline.d.ts +9 -9
  36. package/dist/src/mind/pipeline.js +49 -29
  37. package/dist/src/mind/primitives.d.ts +5 -5
  38. package/dist/src/mind/primitives.js +5 -5
  39. package/dist/src/mind/reasoning.d.ts +5 -1
  40. package/dist/src/mind/reasoning.js +54 -1
  41. package/dist/src/mind/recognition.d.ts +14 -13
  42. package/dist/src/mind/recognition.js +23 -23
  43. package/dist/src/mind/resonance.js +21 -21
  44. package/dist/src/mind/traverse.d.ts +54 -52
  45. package/dist/src/mind/traverse.js +83 -73
  46. package/dist/src/mind/types.d.ts +26 -4
  47. package/dist/src/store.d.ts +12 -12
  48. package/dist/src/store.js +12 -12
  49. package/docs/INDEX.md +2 -2
  50. package/docs/architecture/exact-vs-approximate.md +2 -1
  51. package/docs/architecture/fold-contract.md +1 -1
  52. package/docs/failures/tempting-but-wrong.md +33 -5
  53. package/docs/harness/gates.md +7 -7
  54. package/example/train_base/config.ts +2 -2
  55. package/example/train_base/corpora/massive.ts +1 -1
  56. package/example/train_base/readers.ts +1 -1
  57. package/jsr.json +1 -1
  58. package/package.json +1 -1
  59. package/src/config.ts +35 -0
  60. package/src/geometry.ts +25 -24
  61. package/src/meter.ts +61 -14
  62. package/src/mind/attention.ts +12 -12
  63. package/src/mind/bridge.ts +33 -32
  64. package/src/mind/corpus.ts +202 -0
  65. package/src/mind/graph-search.ts +261 -31
  66. package/src/mind/index.ts +8 -1
  67. package/src/mind/junction.ts +8 -8
  68. package/src/mind/learning.ts +36 -35
  69. package/src/mind/match.ts +163 -73
  70. package/src/mind/mechanisms/cast.ts +17 -1
  71. package/src/mind/mechanisms/cover.ts +18 -12
  72. package/src/mind/mechanisms/prefix-completion.ts +24 -24
  73. package/src/mind/mechanisms/recall.ts +38 -40
  74. package/src/mind/mechanisms/reference.ts +16 -16
  75. package/src/mind/mind.ts +129 -7
  76. package/src/mind/pipeline-mechanism.ts +25 -21
  77. package/src/mind/pipeline.ts +63 -38
  78. package/src/mind/primitives.ts +5 -5
  79. package/src/mind/reasoning.ts +55 -0
  80. package/src/mind/recognition.ts +23 -23
  81. package/src/mind/resonance.ts +21 -21
  82. package/src/mind/traverse.ts +83 -73
  83. package/src/mind/types.ts +30 -4
  84. package/src/store.ts +20 -20
  85. package/test/08-storage.test.mjs +1 -1
  86. package/test/100-complete-grounding-trace.test.mjs +109 -0
  87. package/test/101-alignment-gap-bound.test.mjs +106 -0
  88. package/test/102-production-composes-at-scale.test.mjs +110 -0
  89. package/test/103-alignment-gap-budget.test.mjs +89 -0
  90. package/test/104-composition-is-reported.test.mjs +90 -0
  91. package/test/105-derive-through-reports-its-refusal.test.mjs +113 -0
  92. package/test/106-the-join-fires.test.mjs +94 -0
  93. package/test/107-the-join-is-counted.test.mjs +81 -0
  94. package/test/108-the-join-chains.test.mjs +78 -0
  95. package/test/109-the-pivot-is-counted.test.mjs +60 -0
  96. package/test/110-the-reasoner-stops-when-the-question-is-answered.test.mjs +91 -0
  97. package/test/111-the-cover-assembly-is-counted.test.mjs +74 -0
  98. package/test/112-the-exploration-does-not-grow-with-the-hub.test.mjs +89 -0
  99. package/test/113-the-rationale-payload-is-bounded.test.mjs +84 -0
  100. package/test/114-alignment-budget-is-per-sweep.test.mjs +93 -0
  101. package/test/116-the-extension-is-gated-by-the-pipelines-own-remainder.test.mjs +100 -0
  102. package/test/117-corpus-search.test.mjs +171 -0
  103. package/test/14-scaling.test.mjs +10 -7
  104. package/test/35-prefix-edge.test.mjs +1 -1
  105. package/test/40-choosenext-scale-guard.test.mjs +16 -17
  106. package/test/56-bridge-identity-admission.test.mjs +6 -6
  107. package/test/70-prefix-completion.test.mjs +4 -3
  108. package/test/72-prefix-candidate-supply.test.mjs +3 -3
  109. package/test/73-scaffolding-only-bridge-abstains.test.mjs +6 -6
  110. package/test/75-multiturn-context-optimisation.test.mjs +5 -5
  111. package/test/76-reference-binding.test.mjs +6 -1
  112. package/test/84-composed-answer-honesty.test.mjs +5 -6
  113. package/test/88-dependency-footprint.test.mjs +1 -1
  114. package/test/89-completion-recursion.test.mjs +47 -19
  115. package/test/90-connector-read-cap.test.mjs +10 -8
  116. package/test/93-regime-prediction.test.mjs +10 -10
  117. package/test/94-cross-region-budget.test.mjs +2 -2
  118. package/test/95-wide-resonance-removed.test.mjs +8 -7
  119. package/test/96-bytes-walk-termination.test.mjs +3 -3
@@ -93,13 +93,13 @@ export async function think(ctx, query, mechs) {
93
93
  // ── Pre-computation ──────────────────────────────────────────────────
94
94
  const mechanisms = mechs ?? defaultMechanisms;
95
95
  const meter = ctx.meter;
96
- // recognition is a shared analysis (§2.14 contract 5): it does the query's
96
+ // recognition is a shared analysis (meter.md contract 5): it does the query's
97
97
  // own store work (perceive → foldTree → resolve), which used to land in
98
98
  // `think` and in nothing narrower — the meter's one accounting surface must
99
99
  // charge it to itself, exactly as attention/weave/resonance are charged.
100
- // SYNCHRONOUS phase: recognition is on the sync side of §2.10's seam, so it
101
- // is timed with `timeSync` — wrapping it in a promise would make a profiled
102
- // response await where an unprofiled one does not.
100
+ // SYNCHRONOUS phase: recognition is on the sync side of meter.md's seam, so
101
+ // it is timed with `timeSync` — wrapping it in a promise would make a
102
+ // profiled response await where an unprofiled one does not.
103
103
  const rec = meter
104
104
  ? meter.timeSync("recognise", () => recognise(ctx, query))
105
105
  : recognise(ctx, query);
@@ -113,15 +113,15 @@ export async function think(ctx, query, mechs) {
113
113
  ctx.trace?.step("evalComputation", [rItem(query.subarray(u.i, u.j), "expression", undefined, [u.i, u.j])], [rItem(u.bytes, "result", resolve(ctx, u.bytes) ?? undefined)], "evaluate the recognised operation to its authoritative result");
114
114
  }
115
115
  }
116
- // Phase 2: the shared pre-computation container. Eager fields only
117
- // (recognition, computed spans, guide) — every expensive analysis
118
- // (consensus climb, weave, span-shape classification) is a lazily-cached
119
- // method on Precomputed, first-touched by whichever mechanism's floor
120
- // survives its cheap gates and the worthRunning check. A query no
121
- // mechanism climbs for (e.g. one an extension decided) never climbs.
122
- // NOT phased: the constructor itself is trivial (it only derives `k`), so a
123
- // phase here would add a zero-work entry to every profiled report — the meter
124
- // attributes WORK (§2.14); the trace already represents structure.
116
+ // Phase 2: the shared pre-computation container. Eager fields only
117
+ // (recognition, computed spans, guide) — every expensive analysis (consensus
118
+ // climb, weave, span-shape classification) is a lazily-cached method on
119
+ // Precomputed, first-touched by whichever mechanism's floor survives its
120
+ // cheap gates and the worthRunning check. A query no mechanism climbs for
121
+ // (e.g. one an extension decided) never climbs. NOT phased: the constructor
122
+ // itself is trivial (it only derives `k`), so a phase here would add a
123
+ // zero-work entry to every profiled report — the meter attributes WORK
124
+ // (meter.md); the trace already represents structure.
125
125
  const pre = new Precomputed(ctx, query, rec, computed, ctx._edgeGuide);
126
126
  const grade = (w) => Math.floor(w / STEP);
127
127
  const unaccounted = (spans) => unexplainedSpans(query.length, spans)
@@ -166,16 +166,17 @@ export async function think(ctx, query, mechs) {
166
166
  best = c;
167
167
  };
168
168
  const worthRunning = (floor) => best === null || grade(floor) < grade(best.weight);
169
- // REGIME PREDICTION (R8) — observational only. Once the FIRST mechanism has
170
- // had its turn (cover, which §2.6 places first and floors at 0), the market's
171
- // outcome is already determined by the one cost ladder: the consensus climb
172
- // runs exactly when `worthRunning(2 * STEP)` is true — CAST (floor 2·STEP) is
173
- // the cheapest mechanism that first-touches it, so an incumbent at or below
174
- // grade 2 prunes CAST and, with it, confluence (3·STEP) and extraction
175
- // (CONCEPT+STEP) (retrieval); anything above — or no incumbent — runs the
176
- // full market and the climb (composition). The predicate is `worthRunning`,
177
- // the same function the loop itself uses — nothing is computed here that the
178
- // engine had not already computed, and nothing is read back by inference.
169
+ // REGIME PREDICTION (R8) — observational only. Once the FIRST mechanism has
170
+ // had its turn (cover, which mechanism-market.md places first and floors at
171
+ // 0), the market's outcome is already determined by the one cost ladder: the
172
+ // consensus climb runs exactly when `worthRunning(2 * STEP)` is true — CAST
173
+ // (floor 2·STEP) is the cheapest mechanism that first-touches it, so an
174
+ // incumbent at or below grade 2 prunes CAST and, with it, confluence (3·STEP)
175
+ // and extraction (CONCEPT+STEP) (retrieval); anything above — or no incumbent
176
+ // — runs the full market and the climb (composition). The predicate is
177
+ // `worthRunning`, the same function the loop itself uses — nothing is
178
+ // computed here that the engine had not already computed, and nothing is read
179
+ // back by inference.
179
180
  //
180
181
  // EMITTED BEFORE THE SECOND MECHANISM'S FLOOR, never after some mechanism's
181
182
  // run: a "prediction" published after the fact could assert "the climb will
@@ -354,9 +355,32 @@ export async function think(ctx, query, mechs) {
354
355
  const voiced = (provenance === "cast" || provenance === "join")
355
356
  ? [...castUsed].flatMap((id) => ctx.store.nextFirst(id, hubBound(ctx)).map((n) => read(ctx, n)))
356
357
  : [];
358
+ // REPORTABLE, NOT SILENT. A declared-complete grounding ends the derivation
359
+ // here, and that decision is part of the derivation's shape: the reader of a
360
+ // rationale must be able to see that the chain stopped because the mechanism
361
+ // claimed the query WAS the context, not because nothing followed. The step
362
+ // carries the claim, not a re-description of the answer — the extension is
363
+ // skipped, so there is no output item to show.
364
+ if (decided.complete) {
365
+ ctx.trace?.step("completeGrounding", [rItem(answer, provenance)], [], "grounding declared complete — the query IS the context, so " +
366
+ "post-grounding extension is skipped");
367
+ }
368
+ // THE REASONER JUDGES ITS OWN EXTENSIONS BY THE PIPELINE'S REMAINDER, not by
369
+ // the ladder's `accounted` — and by the SAME reading the fuse gate below uses,
370
+ // with the same W floor. `accounted` is a COST quantity (measured: a query
371
+ // fully explained by one computed span plus bridged connectors reports
372
+ // `accounted: []` while nothing is unexplained), and a remainder under one
373
+ // river-fold quantum is bridging punctuation, never a second topic — so it
374
+ // licenses no extension and blocks none.
375
+ const explained = [
376
+ ...decided.accounted,
377
+ ...pre.computed.map((u) => [u.i, u.j]),
378
+ ];
379
+ const uncovered = unexplainedSpans(query.length, explained)
380
+ .filter(([a, b]) => b - a >= ctx.space.maxGroup);
357
381
  const reasoned = decided.complete ? answer : meter
358
- ? await meter.time("reason", () => reason(ctx, query, answer, preConsumed, pre, voiced))
359
- : await reason(ctx, query, answer, preConsumed, pre, voiced);
382
+ ? await meter.time("reason", () => reason(ctx, query, answer, preConsumed, pre, voiced, uncovered))
383
+ : await reason(ctx, query, answer, preConsumed, pre, voiced, uncovered);
360
384
  // Fuse only when the query has a genuine REMAINDER no mechanism's
361
385
  // structural evidence touched at all. `decided.accounted` alone
362
386
  // undercounts this: it is a COST-LADDER quantity (cover.ts prices its
@@ -373,10 +397,6 @@ export async function think(ctx, query, mechs) {
373
397
  // observed: a single space between two fully-computed arithmetic spans
374
398
  // ("2+2 3+3") registered as "unaccounted" and pulled in an unrelated
375
399
  // corpus fact, corrupting "4 6" into "4 63".
376
- const explained = [
377
- ...decided.accounted,
378
- ...pre.computed.map((u) => [u.i, u.j]),
379
- ];
380
400
  const remainder = unaccounted(explained);
381
401
  // Whether the winning candidate's entire recognised substance is
382
402
  // COMPUTED — every accounted span exactly a pre.computed span, nothing
@@ -20,11 +20,11 @@ export declare function perceiveKey(bytes: Uint8Array, boundaries?: readonly num
20
20
  /** Perceive input into a content-defined tree (the river fold).
21
21
  * Deterministic — identical bytes always produce an identical tree.
22
22
  *
23
- * `boundaries` is an optional sorted list of proper byte offsets where the
24
- * fold must split so that each prefix segment folds identically to how it
25
- * folded when it was learned (§10.3 stable-prefix contract). Only the
26
- * CALLER — who assembled the multi-turn context — knows where those
27
- * boundaries are; the geometry never guesses them from the bytes. */
23
+ * `boundaries` is an optional sorted list of proper byte offsets where the fold
24
+ * must split so that each prefix segment folds identically to how it folded
25
+ * when it was learned (fold-contract.md stable-prefix contract). Only the
26
+ * CALLER — who assembled the multi-turn context — knows where those boundaries
27
+ * are; the geometry never guesses them from the bytes. */
28
28
  export declare function perceive(ctx: MindContext, input: Input, leafAt?: (i: number) => number | null, lookup?: (ids: number[]) => number | null, boundaries?: readonly number[]): Sema;
29
29
  /** The DEPOSIT-shaped perceive. Folds over the stream's own content cuts —
30
30
  * bit-identical to what inference computes for the same bytes. That
@@ -40,11 +40,11 @@ export function perceiveKey(bytes, boundaries) {
40
40
  /** Perceive input into a content-defined tree (the river fold).
41
41
  * Deterministic — identical bytes always produce an identical tree.
42
42
  *
43
- * `boundaries` is an optional sorted list of proper byte offsets where the
44
- * fold must split so that each prefix segment folds identically to how it
45
- * folded when it was learned (§10.3 stable-prefix contract). Only the
46
- * CALLER — who assembled the multi-turn context — knows where those
47
- * boundaries are; the geometry never guesses them from the bytes. */
43
+ * `boundaries` is an optional sorted list of proper byte offsets where the fold
44
+ * must split so that each prefix segment folds identically to how it folded
45
+ * when it was learned (fold-contract.md stable-prefix contract). Only the
46
+ * CALLER — who assembled the multi-turn context — knows where those boundaries
47
+ * are; the geometry never guesses them from the bytes. */
48
48
  export function perceive(ctx, input, leafAt, lookup, boundaries) {
49
49
  if (typeof input === "string" || input instanceof Uint8Array) {
50
50
  const bytes = typeof input === "string"
@@ -20,7 +20,11 @@ export declare function restatesQuery(query: Uint8Array, bytes: Uint8Array): boo
20
20
  * when it declared one — see the pivot's own containment rule. `pre` is the
21
21
  * response's shared pre-computation — the post-grounding stages read the
22
22
  * same container the mechanisms did. */
23
- export declare function reason(ctx: MindContext, query: Uint8Array, answer: Uint8Array, preConsumed: ReadonlySet<number>, pre: Precomputed, voiced?: readonly Uint8Array[]): Promise<Uint8Array>;
23
+ export declare function reason(ctx: MindContext, query: Uint8Array, answer: Uint8Array, preConsumed: ReadonlySet<number>, pre: Precomputed, voiced?: readonly Uint8Array[],
24
+ /** The query material the GROUNDING left uncovered — the cost ladder's own
25
+ * `unaccounted` spans. Only the reasoner's OWN extensions are judged
26
+ * against it; a mechanism carrying its own `used` set owns its shape. */
27
+ uncovered?: readonly (readonly [number, number])[]): Promise<Uint8Array>;
24
28
  /** Fuse independent points of attention into one answer (multi-topic).
25
29
  * When the consensus climb finds more than one dominant point, each
26
30
  * independent point grounds its own answer; they are bridged together
@@ -30,7 +30,11 @@ export function restatesQuery(query, bytes) {
30
30
  * when it declared one — see the pivot's own containment rule. `pre` is the
31
31
  * response's shared pre-computation — the post-grounding stages read the
32
32
  * same container the mechanisms did. */
33
- export async function reason(ctx, query, answer, preConsumed, pre, voiced = []) {
33
+ export async function reason(ctx, query, answer, preConsumed, pre, voiced = [],
34
+ /** The query material the GROUNDING left uncovered — the cost ladder's own
35
+ * `unaccounted` spans. Only the reasoner's OWN extensions are judged
36
+ * against it; a mechanism carrying its own `used` set owns its shape. */
37
+ uncovered = []) {
34
38
  // Echo guard: a query that is ITSELF a learnt continuation (some context's
35
39
  // answer) is being asked back at the system — hopping forward from it would
36
40
  // chain through the very fact that produced it and echo the conversation
@@ -178,6 +182,55 @@ export async function reason(ctx, query, answer, preConsumed, pre, voiced = [])
178
182
  consumeAll(pivot);
179
183
  if (fc === null || bytesEqual(fc, cur) || restatesQuery(query, fc))
180
184
  break;
185
+ // WHOSE EXTENSION IS THIS?
186
+ //
187
+ // `voiced` is what the mechanism WITHHELD (the pipeline sends the used
188
+ // anchors' CONTINUATIONS, not their bytes — see pipeline's own note), so a
189
+ // non-empty `voiced` means exactly what that note says: the grounding came
190
+ // from a mechanism that carries its own short `used` set (cast/join) and
191
+ // therefore owns the shape of its answer. The further terms inside such a
192
+ // seat are legitimately followable — test/29 C3's `Mona Lisa` lives inside
193
+ // the voiced seat and leads on to a fact about neither analog.
194
+ //
195
+ // Every other grounding is ordinary, and an extension of it is the
196
+ // reasoner's own inference: it is taken only while question material the
197
+ // grounding left uncovered remains AND the step carries some of it, judged
198
+ // by the mind's own line between chance and evidence — one W-byte window,
199
+ // no word notion, no character class, no threshold. Measured: the drift's
200
+ // second step (`the Eiffel Tower is in Paris` after `Paris is famous for
201
+ // the Eiffel Tower`) carries no window of `" famous for"` and is refused,
202
+ // while the first carries it. Terminates by a real argument: the uncovered
203
+ // material is finite and each taken extension must carry some of it.
204
+ const producerOwnsShape = voiced.length > 0;
205
+ if (!producerOwnsShape && uncovered.length > 0) {
206
+ const W = ctx.space.maxGroup;
207
+ let progress = false;
208
+ for (const [a, b] of uncovered) {
209
+ for (let i = a; i + W <= b && !progress; i++) {
210
+ if (indexOf(fc, query.subarray(i, i + W), 0) >= 0)
211
+ progress = true;
212
+ }
213
+ if (progress)
214
+ break;
215
+ }
216
+ if (!progress) {
217
+ // THE BRAKE, MADE VISIBLE. The reasoner declines a step that carries
218
+ // none of the material the grounding left uncovered — the drift the
219
+ // extension tests pin. A refusal that leaves no trace is the kind of
220
+ // silent cut AGENTS §6 forbids: the rationale is where a reader learns
221
+ // that an extension was declined for want of question material, and
222
+ // where the next person sees why the chain stopped here. Measured with
223
+ // the check disabled, test/110 and test/116 fail — so this brake is the
224
+ // only thing keeping the extension honest until the pivot reports its
225
+ // own accounted spans and the ladder can judge it instead.
226
+ const left = uncovered.reduce((n, [a, b]) => n + (b - a), 0);
227
+ ctx.trace?.step("pivotRefused", [rItem(cur, "answer"), rItem(query, "query")], uncovered.map(([a, b]) => rItem(query.subarray(a, b), "uncovered")), `the step carries none of the question material the grounding left ` +
228
+ `uncovered (${left} byte(s) in ${uncovered.length} span(s)) — refused`);
229
+ break;
230
+ }
231
+ }
232
+ if (ctx.meter)
233
+ ctx.meter.pivotSteps++;
181
234
  t ??= ctx.trace?.enter("reason", [rItem(startedFrom, "grounded")]);
182
235
  ctx.trace?.step("pivotStep", [rItem(cur, "answer"), rNode(ctx, pivot, "pivot")], [rItem(fc, "answer", resolve(ctx, fc) ?? undefined)], "pivot on the shared span this answer contains, then step forward across that fact");
183
236
  cur = fc;
@@ -12,19 +12,20 @@ import type { MindContext, Recognition, Segment } from "./types.js";
12
12
  *
13
13
  * Both O(n · maxGroup) bounded O(1) probes — never a scan of the corpus.
14
14
  *
15
- * ONE READING PER BYTE STREAM, deliberately: there is no "cheap mode" that
16
- * skips the edge-trim fallbacks. A `trimmed` variant was tried and REFUTED
17
- * twice over. Its premise — "the trims only recover misaligned FRAGMENTS, so
18
- * a consumer whose gate rejects fragments loses nothing" — is false: the
19
- * left/right trim loops below exist precisely to find WHOLE trained forms
20
- * embedded at an offset the query's own fold did not cut, and such a form has
21
- * no structural parents or containers, so it passes the pivot's fragment gate
22
- * and is exactly the candidate a multi-hop chain steps through. Skipping them
23
- * narrows the pivot's evidence silently. And a per-caller variant has to key
24
- * the memo by the variant, which breaks the "computed at most once" property
25
- * (§2.11): the pipeline recognises a grounded answer untrimmed for
26
- * `preConsumed`, and the pivot then recognises the same bytes again — the
27
- * saving inverts into a doubling on the path it was measured for. */
15
+ * ONE READING PER BYTE STREAM, deliberately: there is no "cheap mode" that
16
+ * skips
17
+ * the edge-trim fallbacks. A `trimmed` variant was tried and REFUTED twice
18
+ * over. Its premise — "the trims only recover misaligned FRAGMENTS, so a
19
+ * consumer whose gate rejects fragments loses nothing" — is false: the
20
+ * left/right trim loops below exist precisely to find WHOLE trained forms
21
+ * embedded at an offset the query's own fold did not cut, and such a form has
22
+ * no structural parents or containers, so it passes the pivot's fragment gate
23
+ * and is exactly the candidate a multi-hop chain steps through. Skipping them
24
+ * narrows the pivot's evidence silently. And a per-caller variant has to key
25
+ * the memo by the variant, which breaks the "computed at most once" property
26
+ * (memoization.md): the pipeline recognises a grounded answer untrimmed for
27
+ * `preConsumed`, and the pivot then recognises the same bytes again — the
28
+ * saving inverts into a doubling on the path it was measured for. */
28
29
  export declare function recognise(ctx: MindContext, bytes: Uint8Array): Recognition;
29
30
  /** Segment bytes using the geometry's own groupings — leaf-parent
30
31
  * nodes from the perceived tree, with consecutive bare leaves merged
@@ -23,19 +23,20 @@ import { isChunk } from "../sema.js";
23
23
  *
24
24
  * Both O(n · maxGroup) bounded O(1) probes — never a scan of the corpus.
25
25
  *
26
- * ONE READING PER BYTE STREAM, deliberately: there is no "cheap mode" that
27
- * skips the edge-trim fallbacks. A `trimmed` variant was tried and REFUTED
28
- * twice over. Its premise — "the trims only recover misaligned FRAGMENTS, so
29
- * a consumer whose gate rejects fragments loses nothing" — is false: the
30
- * left/right trim loops below exist precisely to find WHOLE trained forms
31
- * embedded at an offset the query's own fold did not cut, and such a form has
32
- * no structural parents or containers, so it passes the pivot's fragment gate
33
- * and is exactly the candidate a multi-hop chain steps through. Skipping them
34
- * narrows the pivot's evidence silently. And a per-caller variant has to key
35
- * the memo by the variant, which breaks the "computed at most once" property
36
- * (§2.11): the pipeline recognises a grounded answer untrimmed for
37
- * `preConsumed`, and the pivot then recognises the same bytes again — the
38
- * saving inverts into a doubling on the path it was measured for. */
26
+ * ONE READING PER BYTE STREAM, deliberately: there is no "cheap mode" that
27
+ * skips
28
+ * the edge-trim fallbacks. A `trimmed` variant was tried and REFUTED twice
29
+ * over. Its premise — "the trims only recover misaligned FRAGMENTS, so a
30
+ * consumer whose gate rejects fragments loses nothing" — is false: the
31
+ * left/right trim loops below exist precisely to find WHOLE trained forms
32
+ * embedded at an offset the query's own fold did not cut, and such a form has
33
+ * no structural parents or containers, so it passes the pivot's fragment gate
34
+ * and is exactly the candidate a multi-hop chain steps through. Skipping them
35
+ * narrows the pivot's evidence silently. And a per-caller variant has to key
36
+ * the memo by the variant, which breaks the "computed at most once" property
37
+ * (memoization.md): the pipeline recognises a grounded answer untrimmed for
38
+ * `preConsumed`, and the pivot then recognises the same bytes again — the
39
+ * saving inverts into a doubling on the path it was measured for. */
39
40
  export function recognise(ctx, bytes) {
40
41
  // Content-keyed memo — works for both single-turn respond() and multi-turn
41
42
  // respondTurn() (where the map persists across calls). ALWAYS consulted,
@@ -159,16 +160,15 @@ function recogniseImpl(ctx, bytes) {
159
160
  // and read the answer from `starts`, which is exactly {0, W, 2W, …}
160
161
  // because riverFold groups fixed-arity — arithmetic, not evidence.
161
162
  //
162
- // Measured on the 17.9M-node store, over the sites of 7 probes (1 good,
163
- // 11 junk by hand-labelling, corrected for whole-query forms):
164
- // len >= W rejects "hi"(2) "of"(2) "is"(2) "di"(2) "the"(3),
165
- // admits "Eiffel Tower"(12) and both whole-query forms
166
- // len >= W-1 admits "the" — W-1 is the write side's straddle
167
- // neighbour for RETRIEVAL, never a claim about units
168
- // §2.7 saturation admits 11/11 junk: edgeAncestors on a site node
169
- // reaches 1..48 contexts, so dominates(ctx, N) needs
170
- // ctx > 162805 and never fires; every site reads DISC
171
- // rarity does not separate: "hi" has 1 container, "the" 572
163
+ // Measured on the 17.9M-node store, over the sites of 7 probes (1 good, 11
164
+ // junk by hand-labelling, corrected for whole-query forms): len >= W
165
+ // rejects "hi"(2) "of"(2) "is"(2) "di"(2) "the"(3), admits "Eiffel
166
+ // Tower"(12) and both whole-query forms len >= W-1 admits "the" — W-1 is
167
+ // the write side's straddle neighbour for RETRIEVAL, never a claim about
168
+ // units commonality.md saturation admits 11/11 junk: edgeAncestors on a
169
+ // site node reaches 1..48 contexts, so dominates(ctx, N) needs ctx > 162805
170
+ // and never fires; every site reads DISC rarity does not separate: "hi" has
171
+ // 1 container, "the" 572
172
172
  //
173
173
  // A span covering the WHOLE query is exempt: then it is not a fragment of
174
174
  // something longer, it is the question ("hi" asked on its own).
@@ -303,15 +303,15 @@ export async function pivotInto(ctx, answer, consumed, voiced = []) {
303
303
  // Byte containment, longest wins — the answer literally contains the
304
304
  // pivot's bytes, and the biggest well-evidenced span is the real pivot.
305
305
  //
306
- // REAL SATURATION, not a hard cap: the score IS the candidate's byte
307
- // length, so the scan is DECIDED the moment the first candidate that passes
308
- // every filter is found in DESCENDING length order — a shorter candidate can
309
- // never outscore it. `contentLen` (the prefix-capped length read, §2.8) is
310
- // the cheap ordering key, and the first-inserted tie-break is made explicit
311
- // (`a.index - b.index`) so equal lengths keep `scored`'s insertion order —
312
- // exactly the tie argmaxBy(strict) used to keep. The bytes of at most ONE
313
- // winning candidate are read; every shorter candidate the probes proposed is
314
- // skipped without reconstruction, where the old argmax read them all.
306
+ // REAL SATURATION, not a hard cap: the score IS the candidate's byte length,
307
+ // so the scan is DECIDED the moment the first candidate that passes every
308
+ // filter is found in DESCENDING length order — a shorter candidate can never
309
+ // outscore it. `contentLen` (the prefix-capped length read, bounded-reads.md)
310
+ // is the cheap ordering key, and the first-inserted tie-break is made
311
+ // explicit (`a.index - b.index`) so equal lengths keep `scored`'s insertion
312
+ // order — exactly the tie argmaxBy(strict) used to keep. The bytes of at most
313
+ // ONE winning candidate are read; every shorter candidate the probes proposed
314
+ // is skipped without reconstruction, where the old argmax read them all.
315
315
  const ranked = [...scored.keys()]
316
316
  .map((id, index) => ({
317
317
  id,
@@ -322,11 +322,11 @@ export async function pivotInto(ctx, answer, consumed, voiced = []) {
322
322
  let pivotId = null;
323
323
  for (const c of ranked) {
324
324
  const id = c.id;
325
- // A ZERO-LENGTH candidate is not a pivot. `argmaxBy(…, 0, strict)` used to
325
+ // A ZERO-LENGTH candidate is not a pivot. `argmaxBy(…, 0, strict)` used to
326
326
  // carry this floor in its threshold argument, and dropping it here would
327
327
  // admit an empty node: `indexOf(answer, <empty>)` returns 0, so every
328
- // filter below passes and the chain would hop through nothing (§2.13 —
329
- // empty bytes are truthy).
328
+ // filter below passes and the chain would hop through nothing
329
+ // (INVARIANTS.md — empty bytes are truthy).
330
330
  if (c.len === 0)
331
331
  continue;
332
332
  // A PIVOT MUST BE A THING THE CORPUS DEPOSITED, NOT A PIECE OF ONE.
@@ -364,15 +364,15 @@ export async function pivotInto(ctx, answer, consumed, voiced = []) {
364
364
  // No constant enters — it is a structural predicate, not a threshold.
365
365
  if (ctx.store.hasParents(id) || ctx.store.hasContainers(id))
366
366
  continue;
367
- // A candidate whose bytes are LONGER than the answer cannot be a
368
- // substring of it — `indexOf` would return −1 regardless. Prune by
369
- // length BEFORE reconstructing the bytes: `read` is an UNCAPPED read
370
- // (AGENTS §2.8), and a resonated context far longer than the answer is
371
- // exactly the candidate that makes it cost a whole deposit's worth of
372
- // reconstruction for a containment test that must fail. `contentLen`
373
- // with the `answer.length + 1` cap is the prefix-capped length read the
374
- // same contract prescribes; the prune is byte-identical to the old
375
- // `indexOf` miss (it returns −1 for a needle longer than the haystack).
367
+ // A candidate whose bytes are LONGER than the answer cannot be a substring
368
+ // of it — `indexOf` would return −1 regardless. Prune by length BEFORE
369
+ // reconstructing the bytes: `read` is an UNCAPPED read (bounded-reads.md),
370
+ // and a resonated context far longer than the answer is exactly the
371
+ // candidate that makes it cost a whole deposit's worth of reconstruction
372
+ // for a containment test that must fail. `contentLen` with the
373
+ // `answer.length + 1` cap is the prefix-capped length read the same
374
+ // contract prescribes; the prune is byte-identical to the old `indexOf`
375
+ // miss (it returns −1 for a needle longer than the haystack).
376
376
  if (c.len > answer.length)
377
377
  continue;
378
378
  const bytes = read(ctx, id);
@@ -2,14 +2,14 @@ import { Vec } from "../vec.js";
2
2
  import type { AncestorReach, MindContext } from "./types.js";
3
3
  /** The reach memo this ask should use — see the note above.
4
4
  *
5
- * A TRACED response always gets a fresh, empty one. `AncestorReach`'s
6
- * `visited`/`maxDepth`/`saturation` fields are populated only when a trace
7
- * is attached, so an entry deposited by an untraced earlier turn would
8
- * silently black out the reach detail of a later traced one; and the trace's
9
- * reach payload is serialised by ITERATING this map, which must therefore
10
- * hold what THIS climb consulted, not the whole conversation's history.
11
- * Consistent with AGENTS §2.11: a traced response is a different machine —
12
- * never benchmark with a trace attached. */
5
+ * A TRACED response always gets a fresh, empty one. `AncestorReach`'s
6
+ * `visited`/`maxDepth`/`saturation` fields are populated only when a trace is
7
+ * attached, so an entry deposited by an untraced earlier turn would silently
8
+ * black out the reach detail of a later traced one; and the trace's reach
9
+ * payload is serialised by ITERATING this map, which must therefore hold what
10
+ * THIS climb consulted, not the whole conversation's history. Consistent with
11
+ * memoization.md: a traced response is a different machine — never benchmark
12
+ * with a trace attached. */
13
13
  export declare function sharedReachMemo(ctx: MindContext): Map<number, AncestorReach>;
14
14
  /** Invalidate every session-lifetime structural read after a write. */
15
15
  export declare function invalidateStructuralCaches(ctx: MindContext): void;
@@ -59,13 +59,14 @@ export declare function atomIsHub(ctx: MindContext, contextCount: number): boole
59
59
  * predicate, and never as a replacement for it. */
60
60
  export declare function bearsEdge(ctx: MindContext, id: number): boolean;
61
61
  /** Whether a node LEADS SOMEWHERE — it bears a continuation edge or a halo.
62
- * The admission predicate recognition filters sites with (HOW_IT_WORKS
63
- * §15.3): a form that leads nowhere contributes nothing to any derivation.
64
- * Runs once per candidate span on the recognition hot path — `hasNext` is
65
- * cached per response (the same flat-branch ids are probed across prefix
66
- * variants by canonicalChunkId). `hasHalo` is not cached: it's a single
67
- * indexed point probe per candidate, and the candidates that reach this
68
- * check have already been filtered by hasNext above in edgeAncestors. */
62
+ * The admission predicate recognition filters sites with (cover.md): a form
63
+ * that
64
+ * leads nowhere contributes nothing to any derivation. Runs once per candidate
65
+ * span on the recognition hot path — `hasNext` is cached per response (the same
66
+ * flat-branch ids are probed across prefix variants by canonicalChunkId).
67
+ * `hasHalo` is not cached: it's a single indexed point probe per candidate, and
68
+ * the candidates that reach this check have already been filtered by hasNext
69
+ * above in edgeAncestors. */
69
70
  export declare function leadsSomewhere(ctx: MindContext, id: number): boolean;
70
71
  /** The structural IDF read of ONE node: how many distinct learnt contexts
71
72
  * its containment/edge climb reaches, or Infinity when it reaches none or
@@ -86,10 +87,10 @@ export declare function corpusN(ctx: MindContext): number;
86
87
  * convention. */
87
88
  export declare function hubBound(ctx: MindContext): number;
88
89
  /** Cap a candidate list at the hub bound √N (insertion order) — the ONE
89
- * fan-out convention every walk and disambiguation uses (see HOW_IT_WORKS
90
- * §8.6). A node connected to more than √N others is a hub whose individual
91
- * connections carry ~no discriminative information; materialising or scoring
92
- * them all would make single decisions scale with the corpus. */
90
+ * fan-out convention every walk and disambiguation uses (see bounded-reads.md).
91
+ * A node connected to more than √N others is a hub whose individual connections
92
+ * carry ~no discriminative information; materialising or scoring them all would
93
+ * make single decisions scale with the corpus. */
93
94
  export declare function hubCap<T>(ctx: MindContext, ids: readonly T[]): readonly T[];
94
95
  /** Whether `descendant` lies within `ancestor`'s subtree — a structural DAG
95
96
  * relation read off the hash-consed `kids` lists, by a bounded explicit-stack
@@ -100,16 +101,16 @@ export declare function contains(ctx: MindContext, ancestor: number, descendant:
100
101
  * the EXACT half's veto on calling them synonyms.
101
102
  *
102
103
  * Halos measure company, and the strongest company any two forms can keep is
103
- * standing next to each other: a question and its answer co-occur in every
104
- * episode that taught the pair, so their halos SHOULD be similar, and on a
105
- * conversational store they are (measured on the CONV fixture: consecutive
106
- * turns at 0.809 against a 0.516 concept threshold). A gate reading halo
107
- * cosine alone therefore reads adjacency as synonymy and revoices an answer
108
- * in the words of the question it answers — "it hangs in madrid" spliced back
109
- * into "where is it kept now". The distributional layer cannot tell the two
110
- * relations apart, because to it they are the same observation; the exact
111
- * half can, for free, because it stored the edge. §4.1's division of labour
112
- * exactly: approximate proposes, exact decides.
104
+ * standing next to each other: a question and its answer co-occur in every
105
+ * episode that taught the pair, so their halos SHOULD be similar, and on a
106
+ * conversational store they are (measured on the CONV fixture: consecutive
107
+ * turns at 0.809 against a 0.516 concept threshold). A gate reading halo cosine
108
+ * alone therefore reads adjacency as synonymy and revoices an answer in the
109
+ * words of the question it answers — "it hangs in madrid" spliced back into
110
+ * "where is it kept now". The distributional layer cannot tell the two
111
+ * relations apart, because to it they are the same observation; the exact half
112
+ * can, for free, because it stored the edge. halo-sketch.md's division of
113
+ * labour exactly: approximate proposes, exact decides.
113
114
  *
114
115
  * Read LIMITed in both directions at the hub bound — a common continuation's
115
116
  * fan-in is corpus-sized, and no single decision may scale with it. */
@@ -162,12 +163,13 @@ export declare function chooseAmong(ctx: MindContext, candidates: readonly numbe
162
163
  * W-window it spells is contained by more places than the hub bound allows,
163
164
  * i.e. the whole query is corpus-global scaffolding.
164
165
  *
165
- * WHAT IT IS FOR. Several mechanisms ground a query through the literal
166
- * spans it did NOT explain, and those spans are the whole of their evidence.
167
- * When every one of them is a hub, the query says nothing the corpus can be
168
- * held to, and grounding it means picking one of thousands of continuations
169
- * it gives no evidence for — a fabrication whatever the answer happens to be.
170
- * Answering with silence there is the honest degradation contract (§2.13).
166
+ * WHAT IT IS FOR. Several mechanisms ground a query through the literal spans
167
+ * it
168
+ * did NOT explain, and those spans are the whole of their evidence. When every
169
+ * one of them is a hub, the query says nothing the corpus can be held to, and
170
+ * grounding it means picking one of thousands of continuations it gives no
171
+ * evidence for — a fabrication whatever the answer happens to be. Answering
172
+ * with silence there is the honest degradation contract (INVARIANTS.md).
171
173
  *
172
174
  * MEASURED SEPARATION (trained store, hubBound 571) — this is categorical,
173
175
  * not marginal, and it is why the predicate lives here rather than being
@@ -184,11 +186,11 @@ export declare function chooseAmong(ctx: MindContext, candidates: readonly numbe
184
186
  * evidence and sit on the SAME side as the correct ones, so this predicate
185
187
  * is not what makes them silent and cannot be credited for them.
186
188
  *
187
- * NO NEW THRESHOLD (§2.2): `hubBound` is the √N reading of "hub" used
188
- * everywhere, and the containment read is clamped to it exactly as every
189
- * other fan-out read is (§2.8). A query with no stored window at all is NOT
190
- * scaffolding-only — it has no evidence either way, and its callers already
191
- * refuse it on their own terms. */
189
+ * NO NEW THRESHOLD (thresholds.md): `hubBound` is the √N reading of "hub" used
190
+ * everywhere, and the containment read is clamped to it exactly as every other
191
+ * fan-out read is (bounded-reads.md). A query with no stored window at all is
192
+ * NOT scaffolding-only — it has no evidence either way, and its callers already
193
+ * refuse it on their own terms. */
192
194
  export declare function allWindowsAreScaffolding(ctx: MindContext, query: Uint8Array): boolean;
193
195
  /** Trained forms the query may OPEN, proposed from the write side's own
194
196
  * leaf-id window index — the supply of last resort for prefix completion.
@@ -214,17 +216,17 @@ export declare function allWindowsAreScaffolding(ctx: MindContext, query: Uint8A
214
216
  * by climbing containment then parents. Nothing is added to the write side;
215
217
  * this reads an index training already built.
216
218
  *
217
- * BOUNDED (§2.8), AND WITH NO NEW THRESHOLD. The window whose containment is
218
- * SMALLEST carries the most evidence, and one saturated at `hubBound` carries
219
- * none — that is the same √N reading of "hub" the rest of the mind uses, not
220
- * a tuned knob. The upward walk spends a budget of `hubBound` nodes and
221
- * fans out by W, so a hub query enumerates nothing and the caller stays
222
- * silent rather than guessing (§2.13). Measured on the trained store: the
223
- * photosynthesis form at a one-byte truncation picks a window with 52
224
- * containers, visits 446 nodes, and yields exactly ONE candidate that
225
- * survives the caller's byte compare — the form itself.
219
+ * BOUNDED (bounded-reads.md), AND WITH NO NEW THRESHOLD. The window whose
220
+ * containment is SMALLEST carries the most evidence, and one saturated at
221
+ * `hubBound` carries none — that is the same √N reading of "hub" the rest of
222
+ * the mind uses, not a tuned knob. The upward walk spends a budget of
223
+ * `hubBound` nodes and fans out by W, so a hub query enumerates nothing and the
224
+ * caller stays silent rather than guessing (INVARIANTS.md). Measured on the
225
+ * trained store: the photosynthesis form at a one-byte truncation picks a
226
+ * window with 52 containers, visits 446 nodes, and yields exactly ONE candidate
227
+ * that survives the caller's byte compare — the form itself.
226
228
  *
227
- * These are PROPOSALS only. Every candidate still faces the byte-exact
228
- * prefix compare and all three guards below, so a wrong proposal costs one
229
- * bounded read and can never be voiced (§2.3). */
229
+ * These are PROPOSALS only. Every candidate still faces the byte-exact prefix
230
+ * compare and all three guards below, so a wrong proposal costs one bounded
231
+ * read and can never be voiced (exact-vs-approximate.md). */
230
232
  export declare function formsOpenedBy(ctx: MindContext, query: Uint8Array): number[];