@hviana/sema 0.8.0 → 0.8.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. package/AGENTS.md +22 -1
  2. package/DATASETS.md +1 -1
  3. package/dist/example/train_base/config.js +2 -2
  4. package/dist/example/train_base/corpora/massive.js +1 -1
  5. package/dist/example/train_base/readers.js +1 -1
  6. package/dist/src/geometry.d.ts +10 -10
  7. package/dist/src/geometry.js +25 -24
  8. package/dist/src/meter.d.ts +4 -12
  9. package/dist/src/meter.js +14 -14
  10. package/dist/src/mind/attention.js +12 -12
  11. package/dist/src/mind/bridge.d.ts +8 -8
  12. package/dist/src/mind/bridge.js +33 -32
  13. package/dist/src/mind/graph-search.d.ts +0 -8
  14. package/dist/src/mind/graph-search.js +9 -8
  15. package/dist/src/mind/junction.d.ts +1 -1
  16. package/dist/src/mind/junction.js +8 -8
  17. package/dist/src/mind/learning.js +36 -35
  18. package/dist/src/mind/match.js +14 -13
  19. package/dist/src/mind/mechanisms/cover.js +13 -12
  20. package/dist/src/mind/mechanisms/prefix-completion.js +24 -24
  21. package/dist/src/mind/mechanisms/recall.js +38 -40
  22. package/dist/src/mind/mechanisms/reference.js +16 -16
  23. package/dist/src/mind/mind.d.ts +6 -7
  24. package/dist/src/mind/pipeline-mechanism.d.ts +10 -8
  25. package/dist/src/mind/pipeline-mechanism.js +25 -21
  26. package/dist/src/mind/pipeline.d.ts +9 -9
  27. package/dist/src/mind/pipeline.js +24 -23
  28. package/dist/src/mind/primitives.d.ts +5 -5
  29. package/dist/src/mind/primitives.js +5 -5
  30. package/dist/src/mind/recognition.d.ts +14 -13
  31. package/dist/src/mind/recognition.js +23 -23
  32. package/dist/src/mind/resonance.js +21 -21
  33. package/dist/src/mind/traverse.d.ts +54 -52
  34. package/dist/src/mind/traverse.js +74 -72
  35. package/dist/src/mind/types.d.ts +4 -4
  36. package/dist/src/store.d.ts +12 -12
  37. package/dist/src/store.js +12 -12
  38. package/docs/INDEX.md +2 -2
  39. package/docs/architecture/exact-vs-approximate.md +2 -1
  40. package/docs/architecture/fold-contract.md +1 -1
  41. package/docs/failures/tempting-but-wrong.md +2 -3
  42. package/docs/harness/gates.md +7 -7
  43. package/example/train_base/config.ts +2 -2
  44. package/example/train_base/corpora/massive.ts +1 -1
  45. package/example/train_base/readers.ts +1 -1
  46. package/jsr.json +1 -1
  47. package/package.json +1 -1
  48. package/src/geometry.ts +25 -24
  49. package/src/meter.ts +14 -14
  50. package/src/mind/attention.ts +12 -12
  51. package/src/mind/bridge.ts +33 -32
  52. package/src/mind/graph-search.ts +9 -8
  53. package/src/mind/junction.ts +8 -8
  54. package/src/mind/learning.ts +36 -35
  55. package/src/mind/match.ts +20 -19
  56. package/src/mind/mechanisms/cover.ts +13 -12
  57. package/src/mind/mechanisms/prefix-completion.ts +24 -24
  58. package/src/mind/mechanisms/recall.ts +38 -40
  59. package/src/mind/mechanisms/reference.ts +16 -16
  60. package/src/mind/mind.ts +6 -7
  61. package/src/mind/pipeline-mechanism.ts +25 -21
  62. package/src/mind/pipeline.ts +33 -32
  63. package/src/mind/primitives.ts +5 -5
  64. package/src/mind/recognition.ts +23 -23
  65. package/src/mind/resonance.ts +21 -21
  66. package/src/mind/traverse.ts +74 -72
  67. package/src/mind/types.ts +4 -4
  68. package/src/store.ts +20 -20
  69. package/test/08-storage.test.mjs +1 -1
  70. package/test/35-prefix-edge.test.mjs +1 -1
  71. package/test/40-choosenext-scale-guard.test.mjs +16 -17
  72. package/test/56-bridge-identity-admission.test.mjs +6 -6
  73. package/test/70-prefix-completion.test.mjs +4 -3
  74. package/test/72-prefix-candidate-supply.test.mjs +3 -3
  75. package/test/73-scaffolding-only-bridge-abstains.test.mjs +6 -6
  76. package/test/75-multiturn-context-optimisation.test.mjs +5 -5
  77. package/test/84-composed-answer-honesty.test.mjs +5 -6
  78. package/test/88-dependency-footprint.test.mjs +1 -1
  79. package/test/89-completion-recursion.test.mjs +17 -14
  80. package/test/90-connector-read-cap.test.mjs +10 -8
  81. package/test/93-regime-prediction.test.mjs +10 -10
  82. package/test/94-cross-region-budget.test.mjs +2 -2
  83. package/test/95-wide-resonance-removed.test.mjs +8 -7
  84. package/test/96-bytes-walk-termination.test.mjs +3 -3
@@ -2,8 +2,8 @@
2
2
  // bytes (Grounding IV).
3
3
  //
4
4
  // This file is a CONFIGURATION of the shared frame reading in match.ts, not a
5
- // pipeline of its own. The three parts it configures live where §2.5 puts
6
- // them and are reachable by any mechanism:
5
+ // pipeline of its own. The three parts it configures live where
6
+ // match-project.md puts them and are reachable by any mechanism:
7
7
  //
8
8
  // matcher Precomputed.frames() — the frame INVENTORY: which ranked
9
9
  // candidates read as instances of the query's own frame, and
@@ -29,10 +29,10 @@
29
29
  // candidate's continuation UNSUBSTITUTED, so admitting a slot-gap there would
30
30
  // voice the corpus's filler for the asker's referent — the misreference
31
31
  // measured live on the trained store ("How do you say 'flurbish' in French?"
32
- // answered "the way to say hello is \"Bonjour\""). Nor is CAST rewired: its
32
+ // answered "the way to say hello is \"Bonjour\""). Nor is CAST rewired: its
33
33
  // frame gate is WEAVE-local while a slot is COHORT-local, and substituting one
34
- // population for the other is the error §2.7 names. The notion is made
35
- // AVAILABLE, never imposed.
34
+ // population for the other is the error commonality.md names. The notion is
35
+ // made AVAILABLE, never imposed.
36
36
 
37
37
  import type { MindContext } from "../types.js";
38
38
  import type { FrameInstance } from "../match.js";
@@ -52,16 +52,16 @@ import { rItem, rNode, traceFail } from "../trace.js";
52
52
  * agrees with nothing, so no carriage is attested — the same "two or no
53
53
  * constituent" reading frame-filler's contentRuns applies.
54
54
  *
55
- * THIS IS ALSO THE MECHANISM'S REACH. Evidence comes from the shared top-k
56
- * resonance, so a frame the corpus instantiates only ONCE within k is not
57
- * reachable here. Measured on the trained store: `How do you say 'flurbish'
58
- * in French?` finds one instance of its frame in the top 24 — the rest are
59
- * `How do you make …`, a different frame — so this abstains and recall's
60
- * scaffolding-dominated tier answers with the CORPUS's filler. That
61
- * misreference is recall's, and widening the supply is not the fix: the
62
- * exhaustive √N list recall's refusal path builds costs hundreds of
63
- * milliseconds and this runs before it. Abstaining on thin evidence is the
64
- * honest reading (§2.13). */
55
+ * THIS IS ALSO THE MECHANISM'S REACH. Evidence comes from the shared top-k
56
+ * resonance, so a frame the corpus instantiates only ONCE within k is not
57
+ * reachable here. Measured on the trained store: `How do you say 'flurbish' in
58
+ * French?` finds one instance of its frame in the top 24 — the rest are `How do
59
+ * you make …`, a different frame — so this abstains and recall's
60
+ * scaffolding-dominated tier answers with the CORPUS's filler. That
61
+ * misreference is recall's, and widening the supply is not the fix: the
62
+ * exhaustive √N list recall's refusal path builds costs hundreds of
63
+ * milliseconds and this runs before it. Abstaining on thin evidence is the
64
+ * honest reading (INVARIANTS.md). */
65
65
  const MIN_INSTANCES = 2;
66
66
 
67
67
  /** THE VOICING GATES — this mechanism's own reading of a pairing, applied here
@@ -135,7 +135,7 @@ function electFrame(
135
135
  let best: FrameInstance[] = [];
136
136
  for (const group of bySignature.values()) {
137
137
  // Ties keep the FIRST group in insertion order, which is resonance rank —
138
- // corpus-determined, like every other tie-break here (§2.1).
138
+ // corpus-determined, like every other tie-break here (determinism.md).
139
139
  if (group.length > best.length) best = group;
140
140
  }
141
141
  return best;
package/src/mind/mind.ts CHANGED
@@ -211,13 +211,12 @@ export interface MindOptions {
211
211
  host: import("../extension.js").ExtensionHost,
212
212
  ) => import("./pipeline-mechanism.js").PipelineMechanism)[];
213
213
  /** Measure the computational usage of every inference call — see
214
- * src/meter.ts. Off by default and free when off (one null check per
215
- * store read); on, each `respond`/`respondTurn` leaves a {@link
216
- * Mind.lastCost} report behind. Counters are deterministic, so two runs
217
- * of the same query on the same store are diffable; the millisecond
218
- * fields are not. Profiling NEVER changes an answer — but note that
219
- * attaching a RATIONALE does (traced responses bypass the ctx memos,
220
- * AGENTS §2.11), so profile without a trace. */
214
+ * src/meter.ts. Off by default and free when off (one null check per store
215
+ * read); on, each `respond`/`respondTurn` leaves a {@link Mind.lastCost}
216
+ * report behind. Counters are deterministic, so two runs of the same query on
217
+ * the same store are diffable; the millisecond fields are not. Profiling
218
+ * NEVER changes an answer — but attaching a RATIONALE does: a traced response
219
+ * bypasses the ctx memos (memoization.md), so profile without a trace. */
221
220
  profile?: boolean;
222
221
  /** Content canonicalizer applied to EVERY response (any modality) for
223
222
  * equivalence-class resolution — see src/canon.ts. Text entry points
@@ -5,7 +5,7 @@
5
5
  // a list of PipelineMechanism objects — it never imports a mechanism-specific
6
6
  // type and never has a special-case branch for any mechanism.
7
7
  //
8
- // The four constraints of the free-will architecture (§14.5):
8
+ // The four constraints of the free-will architecture (mechanism-market.md):
9
9
  // 1. DECOUPLING — mechanisms import nothing from each other or from pipeline.
10
10
  // 2. DECLARED COMPETENCE — floor() returns null when impossible, a number when
11
11
  // possible. Binary, auditable, no learned scores.
@@ -148,17 +148,17 @@ export class Precomputed {
148
148
  );
149
149
  }
150
150
 
151
- // REMOVED — the WIDE exhaustive-√N resonance list (`wideResonance`). It ran
151
+ // REMOVED — the WIDE exhaustive-√N resonance list (`wideResonance`). It ran
152
152
  // `resonate(guide, √N, exhaustive=true)` whenever the top hit cleared
153
- // conceptThreshold, so consumers could look "past the top-k". Every consumer
153
+ // conceptThreshold, so consumers could look "past the top-k". Every consumer
154
154
  // only ever needed ≤ 2·recallQueryK proposals (the substitution bridge's own
155
155
  // candidate cap) or a content-addressed answer (prefix completion's
156
- // formsOpenedBy), and every proposal is byte-verified downstream (§2.3), so
157
- // the exhaustive scan bought recall at O(index) cost for an O(k) need —
158
- // measured: 244K annVectorReads per refusing query, ~1.5 s, every answer
159
- // byte-identical to a top-k read. The two consumers now read `resonance()`
160
- // (the one top-k read) and the write side's window index respectively — see
161
- // recall.ts and prefix-completion.ts.
156
+ // formsOpenedBy), and every proposal is byte-verified downstream
157
+ // (exact-vs-approximate.md), so the exhaustive scan bought recall at O(index)
158
+ // cost for an O(k) need — measured: 244K annVectorReads per refusing query,
159
+ // ~1.5 s, every answer byte-identical to a top-k read. The two consumers now
160
+ // read `resonance()` (the one top-k read) and the write side's window index
161
+ // respectively — see recall.ts and prefix-completion.ts.
162
162
 
163
163
  private _frames?: Promise<ReadonlyArray<FrameInstance>>;
164
164
  /** THE FRAME INVENTORY — every ranked candidate that reads as an instance of
@@ -166,14 +166,16 @@ export class Precomputed {
166
166
  * ({@link FrameInstance}). The one place the engine represents "a position
167
167
  * whose occupant comes from the context rather than the corpus".
168
168
  *
169
- * AN INVENTORY, NOT AN ELECTION. It reports every pairing and elects no
170
- * frame, deliberately: a slot is a property of a PAIRING, not of the query,
171
- * and different candidates put slots in different places. Committing to one
172
- * reading here would push whichever consumer asked first onto everyone else
173
- * — the market's decoupling (§2.6) broken from inside the shared container,
174
- * and the population error §2.7 names. Each consumer groups and commits
175
- * for its own question; reference elects the modal slot signature, and a
176
- * consumer wanting a different reading is not fighting this one.
169
+ * AN INVENTORY, NOT AN ELECTION. It reports every pairing and elects no
170
+ * frame,
171
+ * deliberately: a slot is a property of a PAIRING, not of the query, and
172
+ * different candidates put slots in different places. Committing to one
173
+ * reading here would push whichever consumer asked first onto everyone else —
174
+ * the market's decoupling (mechanism-market.md) broken from inside the shared
175
+ * container, and the population error commonality.md names. Each consumer
176
+ * groups and commits for its own question; reference elects the modal slot
177
+ * signature, and a consumer wanting a different reading is not fighting this
178
+ * one.
177
179
  *
178
180
  * NO LICENCE EITHER. Knowing a span is variable is safe for every consumer
179
181
  * — it can only improve an alignment. Knowing one may be VOICED through is
@@ -189,9 +191,10 @@ export class Precomputed {
189
191
  const capBytes = this.query.length * W;
190
192
  const out: FrameInstance[] = [];
191
193
  for (const h of await this.resonance()) {
192
- // REJECT BY LENGTH BEFORE RECONSTRUCTING (§2.8): `contentLen` is an
193
- // indexed read, `bytesPrefix` rebuilds a subtree. ONLY the phrase-scale
194
- // cap is applied — it is a bounded-read discipline, not a judgement.
194
+ // REJECT BY LENGTH BEFORE RECONSTRUCTING (bounded-reads.md):
195
+ // `contentLen` is an indexed read, `bytesPrefix` rebuilds a subtree.
196
+ // ONLY the phrase-scale cap is applied — it is a bounded-read
197
+ // discipline, not a judgement.
195
198
  //
196
199
  // A LOWER bound was here too (`dominates(len, query.length)`, on the
197
200
  // reasoning that a candidate shorter than half the query cannot supply
@@ -519,7 +522,8 @@ function computeWeave(
519
522
  // IDF — gates the aligner has no equivalent of.
520
523
  //
521
524
  // So the climb PROPOSES the pairing (which structure, which query span) and
522
- // bytes DECIDE its terms (§2.3). Three gates, each one measured:
525
+ // bytes DECIDE its terms (exact-vs-approximate.md). Three gates, each one
526
+ // measured:
523
527
  //
524
528
  // • it may only take query bytes NO literal run claimed. Run inline with
525
529
  // phase 1 this did the opposite of "exact decides" — a higher-ranked
@@ -130,15 +130,15 @@ export interface NarrowDecisionData {
130
130
  }
131
131
 
132
132
  /** Structured payload of the "regimePrediction" rationale step — the R8
133
- * observation exposed as data. After the first mechanism (cover, which §2.6
134
- * runs first) grounds or abstains, the market's whole outcome is already
135
- * determined by the one cost ladder: the consensus climb runs exactly when
136
- * `worthRunning(2 * STEP)` is true — CAST (floor 2·STEP) is the cheapest
137
- * mechanism that first-touches it, and confluence (3·STEP) / extraction
138
- * (CONCEPT+STEP) are only reached after CAST is. An incumbent at or below
139
- * that floor prunes CAST and, with it, the climb (retrieval); anything above
140
- * — or no incumbent — runs the full market and the climb (composition).
141
- * Purely observational; never read by inference. */
133
+ * observation exposed as data. After the first mechanism (cover, which
134
+ * mechanism-market.md runs first) grounds or abstains, the market's whole
135
+ * outcome is already determined by the one cost ladder: the consensus climb
136
+ * runs exactly when `worthRunning(2 * STEP)` is true — CAST (floor 2·STEP) is
137
+ * the cheapest mechanism that first-touches it, and confluence (3·STEP) /
138
+ * extraction (CONCEPT+STEP) are only reached after CAST is. An incumbent at or
139
+ * below that floor prunes CAST and, with it, the climb (retrieval); anything
140
+ * above — or no incumbent — runs the full market and the climb (composition).
141
+ * Purely observational; never read by inference. */
142
142
  export interface RegimePredictionData {
143
143
  version: 1;
144
144
  /** retrieval | composition — the two regimes R1 measured as a ~100× cost
@@ -187,13 +187,13 @@ export async function think(
187
187
  // ── Pre-computation ──────────────────────────────────────────────────
188
188
  const mechanisms = mechs ?? defaultMechanisms;
189
189
  const meter = ctx.meter;
190
- // recognition is a shared analysis (§2.14 contract 5): it does the query's
190
+ // recognition is a shared analysis (meter.md contract 5): it does the query's
191
191
  // own store work (perceive → foldTree → resolve), which used to land in
192
192
  // `think` and in nothing narrower — the meter's one accounting surface must
193
193
  // charge it to itself, exactly as attention/weave/resonance are charged.
194
- // SYNCHRONOUS phase: recognition is on the sync side of §2.10's seam, so it
195
- // is timed with `timeSync` — wrapping it in a promise would make a profiled
196
- // response await where an unprofiled one does not.
194
+ // SYNCHRONOUS phase: recognition is on the sync side of meter.md's seam, so
195
+ // it is timed with `timeSync` — wrapping it in a promise would make a
196
+ // profiled response await where an unprofiled one does not.
197
197
  const rec = meter
198
198
  ? meter.timeSync("recognise", () => recognise(ctx, query))
199
199
  : recognise(ctx, query);
@@ -225,15 +225,15 @@ export async function think(
225
225
  }
226
226
  }
227
227
 
228
- // Phase 2: the shared pre-computation container. Eager fields only
229
- // (recognition, computed spans, guide) — every expensive analysis
230
- // (consensus climb, weave, span-shape classification) is a lazily-cached
231
- // method on Precomputed, first-touched by whichever mechanism's floor
232
- // survives its cheap gates and the worthRunning check. A query no
233
- // mechanism climbs for (e.g. one an extension decided) never climbs.
234
- // NOT phased: the constructor itself is trivial (it only derives `k`), so a
235
- // phase here would add a zero-work entry to every profiled report — the meter
236
- // attributes WORK (§2.14); the trace already represents structure.
228
+ // Phase 2: the shared pre-computation container. Eager fields only
229
+ // (recognition, computed spans, guide) — every expensive analysis (consensus
230
+ // climb, weave, span-shape classification) is a lazily-cached method on
231
+ // Precomputed, first-touched by whichever mechanism's floor survives its
232
+ // cheap gates and the worthRunning check. A query no mechanism climbs for
233
+ // (e.g. one an extension decided) never climbs. NOT phased: the constructor
234
+ // itself is trivial (it only derives `k`), so a phase here would add a
235
+ // zero-work entry to every profiled report — the meter attributes WORK
236
+ // (meter.md); the trace already represents structure.
237
237
  const pre = new Precomputed(ctx, query, rec, computed, ctx._edgeGuide);
238
238
 
239
239
  // ── Grounding: ONE lightest-derivation choice among the mechanisms ────
@@ -298,16 +298,17 @@ export async function think(
298
298
  const worthRunning = (floor: number) =>
299
299
  best === null || grade(floor) < grade(best.weight);
300
300
 
301
- // REGIME PREDICTION (R8) — observational only. Once the FIRST mechanism has
302
- // had its turn (cover, which §2.6 places first and floors at 0), the market's
303
- // outcome is already determined by the one cost ladder: the consensus climb
304
- // runs exactly when `worthRunning(2 * STEP)` is true — CAST (floor 2·STEP) is
305
- // the cheapest mechanism that first-touches it, so an incumbent at or below
306
- // grade 2 prunes CAST and, with it, confluence (3·STEP) and extraction
307
- // (CONCEPT+STEP) (retrieval); anything above — or no incumbent — runs the
308
- // full market and the climb (composition). The predicate is `worthRunning`,
309
- // the same function the loop itself uses — nothing is computed here that the
310
- // engine had not already computed, and nothing is read back by inference.
301
+ // REGIME PREDICTION (R8) — observational only. Once the FIRST mechanism has
302
+ // had its turn (cover, which mechanism-market.md places first and floors at
303
+ // 0), the market's outcome is already determined by the one cost ladder: the
304
+ // consensus climb runs exactly when `worthRunning(2 * STEP)` is true — CAST
305
+ // (floor 2·STEP) is the cheapest mechanism that first-touches it, so an
306
+ // incumbent at or below grade 2 prunes CAST and, with it, confluence (3·STEP)
307
+ // and extraction (CONCEPT+STEP) (retrieval); anything above — or no incumbent
308
+ // — runs the full market and the climb (composition). The predicate is
309
+ // `worthRunning`, the same function the loop itself uses — nothing is
310
+ // computed here that the engine had not already computed, and nothing is read
311
+ // back by inference.
311
312
  //
312
313
  // EMITTED BEFORE THE SECOND MECHANISM'S FLOOR, never after some mechanism's
313
314
  // run: a "prediction" published after the fact could assert "the climb will
@@ -59,11 +59,11 @@ export function perceiveKey(
59
59
  /** Perceive input into a content-defined tree (the river fold).
60
60
  * Deterministic — identical bytes always produce an identical tree.
61
61
  *
62
- * `boundaries` is an optional sorted list of proper byte offsets where the
63
- * fold must split so that each prefix segment folds identically to how it
64
- * folded when it was learned (§10.3 stable-prefix contract). Only the
65
- * CALLER — who assembled the multi-turn context — knows where those
66
- * boundaries are; the geometry never guesses them from the bytes. */
62
+ * `boundaries` is an optional sorted list of proper byte offsets where the fold
63
+ * must split so that each prefix segment folds identically to how it folded
64
+ * when it was learned (fold-contract.md stable-prefix contract). Only the
65
+ * CALLER — who assembled the multi-turn context — knows where those boundaries
66
+ * are; the geometry never guesses them from the bytes. */
67
67
  export function perceive(
68
68
  ctx: MindContext,
69
69
  input: Input,
@@ -34,19 +34,20 @@ import type { Leaf, Site } from "./graph-search.js";
34
34
  *
35
35
  * Both O(n · maxGroup) bounded O(1) probes — never a scan of the corpus.
36
36
  *
37
- * ONE READING PER BYTE STREAM, deliberately: there is no "cheap mode" that
38
- * skips the edge-trim fallbacks. A `trimmed` variant was tried and REFUTED
39
- * twice over. Its premise — "the trims only recover misaligned FRAGMENTS, so
40
- * a consumer whose gate rejects fragments loses nothing" — is false: the
41
- * left/right trim loops below exist precisely to find WHOLE trained forms
42
- * embedded at an offset the query's own fold did not cut, and such a form has
43
- * no structural parents or containers, so it passes the pivot's fragment gate
44
- * and is exactly the candidate a multi-hop chain steps through. Skipping them
45
- * narrows the pivot's evidence silently. And a per-caller variant has to key
46
- * the memo by the variant, which breaks the "computed at most once" property
47
- * (§2.11): the pipeline recognises a grounded answer untrimmed for
48
- * `preConsumed`, and the pivot then recognises the same bytes again — the
49
- * saving inverts into a doubling on the path it was measured for. */
37
+ * ONE READING PER BYTE STREAM, deliberately: there is no "cheap mode" that
38
+ * skips
39
+ * the edge-trim fallbacks. A `trimmed` variant was tried and REFUTED twice
40
+ * over. Its premise — "the trims only recover misaligned FRAGMENTS, so a
41
+ * consumer whose gate rejects fragments loses nothing" — is false: the
42
+ * left/right trim loops below exist precisely to find WHOLE trained forms
43
+ * embedded at an offset the query's own fold did not cut, and such a form has
44
+ * no structural parents or containers, so it passes the pivot's fragment gate
45
+ * and is exactly the candidate a multi-hop chain steps through. Skipping them
46
+ * narrows the pivot's evidence silently. And a per-caller variant has to key
47
+ * the memo by the variant, which breaks the "computed at most once" property
48
+ * (memoization.md): the pipeline recognises a grounded answer untrimmed for
49
+ * `preConsumed`, and the pivot then recognises the same bytes again — the
50
+ * saving inverts into a doubling on the path it was measured for. */
50
51
  export function recognise(
51
52
  ctx: MindContext,
52
53
  bytes: Uint8Array,
@@ -180,16 +181,15 @@ function recogniseImpl(ctx: MindContext, bytes: Uint8Array): Recognition {
180
181
  // and read the answer from `starts`, which is exactly {0, W, 2W, …}
181
182
  // because riverFold groups fixed-arity — arithmetic, not evidence.
182
183
  //
183
- // Measured on the 17.9M-node store, over the sites of 7 probes (1 good,
184
- // 11 junk by hand-labelling, corrected for whole-query forms):
185
- // len >= W rejects "hi"(2) "of"(2) "is"(2) "di"(2) "the"(3),
186
- // admits "Eiffel Tower"(12) and both whole-query forms
187
- // len >= W-1 admits "the" — W-1 is the write side's straddle
188
- // neighbour for RETRIEVAL, never a claim about units
189
- // §2.7 saturation admits 11/11 junk: edgeAncestors on a site node
190
- // reaches 1..48 contexts, so dominates(ctx, N) needs
191
- // ctx > 162805 and never fires; every site reads DISC
192
- // rarity does not separate: "hi" has 1 container, "the" 572
184
+ // Measured on the 17.9M-node store, over the sites of 7 probes (1 good, 11
185
+ // junk by hand-labelling, corrected for whole-query forms): len >= W
186
+ // rejects "hi"(2) "of"(2) "is"(2) "di"(2) "the"(3), admits "Eiffel
187
+ // Tower"(12) and both whole-query forms len >= W-1 admits "the" — W-1 is
188
+ // the write side's straddle neighbour for RETRIEVAL, never a claim about
189
+ // units commonality.md saturation admits 11/11 junk: edgeAncestors on a
190
+ // site node reaches 1..48 contexts, so dominates(ctx, N) needs ctx > 162805
191
+ // and never fires; every site reads DISC rarity does not separate: "hi" has
192
+ // 1 container, "the" 572
193
193
  //
194
194
  // A span covering the WHOLE query is exempt: then it is not a fragment of
195
195
  // something longer, it is the question ("hi" asked on its own).
@@ -380,15 +380,15 @@ export async function pivotInto(
380
380
  // Byte containment, longest wins — the answer literally contains the
381
381
  // pivot's bytes, and the biggest well-evidenced span is the real pivot.
382
382
  //
383
- // REAL SATURATION, not a hard cap: the score IS the candidate's byte
384
- // length, so the scan is DECIDED the moment the first candidate that passes
385
- // every filter is found in DESCENDING length order — a shorter candidate can
386
- // never outscore it. `contentLen` (the prefix-capped length read, §2.8) is
387
- // the cheap ordering key, and the first-inserted tie-break is made explicit
388
- // (`a.index - b.index`) so equal lengths keep `scored`'s insertion order —
389
- // exactly the tie argmaxBy(strict) used to keep. The bytes of at most ONE
390
- // winning candidate are read; every shorter candidate the probes proposed is
391
- // skipped without reconstruction, where the old argmax read them all.
383
+ // REAL SATURATION, not a hard cap: the score IS the candidate's byte length,
384
+ // so the scan is DECIDED the moment the first candidate that passes every
385
+ // filter is found in DESCENDING length order — a shorter candidate can never
386
+ // outscore it. `contentLen` (the prefix-capped length read, bounded-reads.md)
387
+ // is the cheap ordering key, and the first-inserted tie-break is made
388
+ // explicit (`a.index - b.index`) so equal lengths keep `scored`'s insertion
389
+ // order — exactly the tie argmaxBy(strict) used to keep. The bytes of at most
390
+ // ONE winning candidate are read; every shorter candidate the probes proposed
391
+ // is skipped without reconstruction, where the old argmax read them all.
392
392
  const ranked = [...scored.keys()]
393
393
  .map((id, index) => ({
394
394
  id,
@@ -399,11 +399,11 @@ export async function pivotInto(
399
399
  let pivotId: number | null = null;
400
400
  for (const c of ranked) {
401
401
  const id = c.id;
402
- // A ZERO-LENGTH candidate is not a pivot. `argmaxBy(…, 0, strict)` used to
402
+ // A ZERO-LENGTH candidate is not a pivot. `argmaxBy(…, 0, strict)` used to
403
403
  // carry this floor in its threshold argument, and dropping it here would
404
404
  // admit an empty node: `indexOf(answer, <empty>)` returns 0, so every
405
- // filter below passes and the chain would hop through nothing (§2.13 —
406
- // empty bytes are truthy).
405
+ // filter below passes and the chain would hop through nothing
406
+ // (INVARIANTS.md — empty bytes are truthy).
407
407
  if (c.len === 0) continue;
408
408
  // A PIVOT MUST BE A THING THE CORPUS DEPOSITED, NOT A PIECE OF ONE.
409
409
  // "Longest wins" ranks candidates but never asks whether the winner is
@@ -439,15 +439,15 @@ export async function pivotInto(
439
439
  // a span that was never a fact on its own is not one to step through.
440
440
  // No constant enters — it is a structural predicate, not a threshold.
441
441
  if (ctx.store.hasParents(id) || ctx.store.hasContainers(id)) continue;
442
- // A candidate whose bytes are LONGER than the answer cannot be a
443
- // substring of it — `indexOf` would return −1 regardless. Prune by
444
- // length BEFORE reconstructing the bytes: `read` is an UNCAPPED read
445
- // (AGENTS §2.8), and a resonated context far longer than the answer is
446
- // exactly the candidate that makes it cost a whole deposit's worth of
447
- // reconstruction for a containment test that must fail. `contentLen`
448
- // with the `answer.length + 1` cap is the prefix-capped length read the
449
- // same contract prescribes; the prune is byte-identical to the old
450
- // `indexOf` miss (it returns −1 for a needle longer than the haystack).
442
+ // A candidate whose bytes are LONGER than the answer cannot be a substring
443
+ // of it — `indexOf` would return −1 regardless. Prune by length BEFORE
444
+ // reconstructing the bytes: `read` is an UNCAPPED read (bounded-reads.md),
445
+ // and a resonated context far longer than the answer is exactly the
446
+ // candidate that makes it cost a whole deposit's worth of reconstruction
447
+ // for a containment test that must fail. `contentLen` with the
448
+ // `answer.length + 1` cap is the prefix-capped length read the same
449
+ // contract prescribes; the prune is byte-identical to the old `indexOf`
450
+ // miss (it returns −1 for a needle longer than the haystack).
451
451
  if (c.len > answer.length) continue;
452
452
  const bytes = read(ctx, id);
453
453
  if (indexOf(answer, bytes, 0) < 0) continue;