@hviana/sema 0.7.9 → 0.8.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (86) hide show
  1. package/AGENTS.md +22 -1
  2. package/DATASETS.md +1 -1
  3. package/dist/example/train_base/config.js +2 -2
  4. package/dist/example/train_base/corpora/massive.js +1 -1
  5. package/dist/example/train_base/readers.js +1 -1
  6. package/dist/src/geometry.d.ts +10 -10
  7. package/dist/src/geometry.js +25 -24
  8. package/dist/src/meter.d.ts +4 -12
  9. package/dist/src/meter.js +14 -14
  10. package/dist/src/mind/attention.js +12 -12
  11. package/dist/src/mind/bridge.d.ts +8 -8
  12. package/dist/src/mind/bridge.js +33 -32
  13. package/dist/src/mind/graph-search.d.ts +0 -8
  14. package/dist/src/mind/graph-search.js +38 -25
  15. package/dist/src/mind/junction.d.ts +1 -1
  16. package/dist/src/mind/junction.js +8 -8
  17. package/dist/src/mind/learning.js +36 -35
  18. package/dist/src/mind/match.js +14 -13
  19. package/dist/src/mind/mechanisms/cover.js +13 -12
  20. package/dist/src/mind/mechanisms/prefix-completion.js +24 -24
  21. package/dist/src/mind/mechanisms/recall.js +38 -40
  22. package/dist/src/mind/mechanisms/reference.js +16 -16
  23. package/dist/src/mind/mind.d.ts +6 -7
  24. package/dist/src/mind/pipeline-mechanism.d.ts +10 -8
  25. package/dist/src/mind/pipeline-mechanism.js +25 -21
  26. package/dist/src/mind/pipeline.d.ts +9 -9
  27. package/dist/src/mind/pipeline.js +24 -23
  28. package/dist/src/mind/primitives.d.ts +5 -5
  29. package/dist/src/mind/primitives.js +5 -5
  30. package/dist/src/mind/recognition.d.ts +14 -13
  31. package/dist/src/mind/recognition.js +53 -38
  32. package/dist/src/mind/resonance.js +21 -21
  33. package/dist/src/mind/traverse.d.ts +54 -52
  34. package/dist/src/mind/traverse.js +74 -72
  35. package/dist/src/mind/types.d.ts +4 -4
  36. package/dist/src/store.d.ts +12 -12
  37. package/dist/src/store.js +12 -12
  38. package/docs/INDEX.md +2 -2
  39. package/docs/architecture/exact-vs-approximate.md +2 -1
  40. package/docs/architecture/fold-contract.md +1 -1
  41. package/docs/failures/tempting-but-wrong.md +2 -3
  42. package/docs/harness/gates.md +7 -7
  43. package/example/train_base/config.ts +2 -2
  44. package/example/train_base/corpora/massive.ts +1 -1
  45. package/example/train_base/readers.ts +1 -1
  46. package/jsr.json +1 -1
  47. package/package.json +1 -1
  48. package/src/geometry.ts +25 -24
  49. package/src/meter.ts +14 -14
  50. package/src/mind/attention.ts +12 -12
  51. package/src/mind/bridge.ts +33 -32
  52. package/src/mind/graph-search.ts +43 -24
  53. package/src/mind/junction.ts +8 -8
  54. package/src/mind/learning.ts +36 -35
  55. package/src/mind/match.ts +20 -19
  56. package/src/mind/mechanisms/cover.ts +13 -12
  57. package/src/mind/mechanisms/prefix-completion.ts +24 -24
  58. package/src/mind/mechanisms/recall.ts +38 -40
  59. package/src/mind/mechanisms/reference.ts +16 -16
  60. package/src/mind/mind.ts +6 -7
  61. package/src/mind/pipeline-mechanism.ts +25 -21
  62. package/src/mind/pipeline.ts +33 -32
  63. package/src/mind/primitives.ts +5 -5
  64. package/src/mind/recognition.ts +51 -36
  65. package/src/mind/resonance.ts +21 -21
  66. package/src/mind/traverse.ts +74 -72
  67. package/src/mind/types.ts +4 -4
  68. package/src/store.ts +20 -20
  69. package/test/08-storage.test.mjs +1 -1
  70. package/test/35-prefix-edge.test.mjs +1 -1
  71. package/test/40-choosenext-scale-guard.test.mjs +16 -17
  72. package/test/46-recognise-multibyte-edge.test.mjs +33 -0
  73. package/test/56-bridge-identity-admission.test.mjs +6 -6
  74. package/test/70-prefix-completion.test.mjs +4 -3
  75. package/test/72-prefix-candidate-supply.test.mjs +3 -3
  76. package/test/73-scaffolding-only-bridge-abstains.test.mjs +6 -6
  77. package/test/75-multiturn-context-optimisation.test.mjs +5 -5
  78. package/test/84-composed-answer-honesty.test.mjs +5 -6
  79. package/test/88-dependency-footprint.test.mjs +1 -1
  80. package/test/89-completion-recursion.test.mjs +17 -14
  81. package/test/90-connector-read-cap.test.mjs +10 -8
  82. package/test/93-regime-prediction.test.mjs +10 -10
  83. package/test/94-cross-region-budget.test.mjs +2 -2
  84. package/test/95-wide-resonance-removed.test.mjs +8 -7
  85. package/test/96-bytes-walk-termination.test.mjs +3 -3
  86. package/test/99-fact-join.test.mjs +38 -0
@@ -236,24 +236,22 @@ export async function recallByResonance(
236
236
  }
237
237
  }
238
238
 
239
- // The query-relative grounding fraction, shared by tiers 2–4 — gated on
240
- // the FRACTION OF THE QUERY the grounding explains, not the raw cosine.
241
- // Root gists are unit vectors, but their magnitudes are recoverable from
242
- // the byte lengths (‖·‖ = √len under the linear fold):
243
- // cos = shared/√(lenQ·lenG), so shared/lenQ = cos·√(lenG/lenQ).
244
- // The raw cosine punished honest containment — a query fully inside a
245
- // longer grounded answer scored √(lenQ/lenG) and was refused — and let a
246
- // long answer sharing only scaffolding pass; the query-relative fraction
247
- // measures exactly what the reach bar means: how much of THE QUERY the
248
- // store accounts for.
249
- // Chance similarity survives the length conversion AMPLIFIED: the same
250
- // √(lenG/lenQ) factor that converts an honest shared fraction into a
251
- // query-relative one multiplies the estimator/chance floor too, so a long
252
- // stored form (√(lenG/lenQ) ≈ 10 at 100×) lifted a noise-level cosine past
253
- // the reach bar and grounded pure gibberish (observed). Only the
254
- // ABOVE-CHANCE part of the similarity is evidence of shared content —
255
- // subtract the significance bar (3/√D, §8.3) before converting. Derived
256
- // from the existing bars; never tuned.
239
+ // The query-relative grounding fraction, shared by tiers 2–4 — gated on the
240
+ // FRACTION OF THE QUERY the grounding explains, not the raw cosine. Root
241
+ // gists are unit vectors, but their magnitudes are recoverable from the byte
242
+ // lengths (‖·‖ = √len under the linear fold): cos = shared/√(lenQ·lenG), so
243
+ // shared/lenQ = cos·√(lenG/lenQ). The raw cosine punished honest containment
244
+ // — a query fully inside a longer grounded answer scored √(lenQ/lenG) and was
245
+ // refused — and let a long answer sharing only scaffolding pass; the
246
+ // query-relative fraction measures exactly what the reach bar means: how much
247
+ // of THE QUERY the store accounts for. Chance similarity survives the length
248
+ // conversion AMPLIFIED: the same √(lenG/lenQ) factor that converts an honest
249
+ // shared fraction into a query-relative one multiplies the estimator/chance
250
+ // floor too, so a long stored form (√(lenG/lenQ) ≈ 10 at 100×) lifted a
251
+ // noise-level cosine past the reach bar and grounded pure gibberish
252
+ // (observed). Only the ABOVE-CHANCE part of the similarity is evidence of
253
+ // shared content — subtract the significance bar (3/√D, thresholds.md) before
254
+ // converting. Derived from the existing bars; never tuned.
257
255
  const sig = significanceBar(ctx.store.D);
258
256
  const reach = reachThreshold(ctx.space.maxGroup);
259
257
  const fracOfQuery = (cos: number, otherLen: number): number =>
@@ -411,14 +409,14 @@ export async function recallByResonance(
411
409
  }
412
410
  }
413
411
  }
414
- // 3b. Corroborated-substitution bridge — refusal-path only (bridge.ts).
415
- // The bridge's proposal source is the response's ONE top-k read — the same
416
- // list recall already ranked above — never an exhaustive √N scan. The
417
- // bridge's own candidate cap is 2·recallQueryK, so top-k proposals are
418
- // exactly the budget it can consume, and every proposal is byte-verified
419
- // downstream (§2.3). Reuse the memoised `resonance()`; scanning every IVF
420
- // cluster here once made every honest refusal cost hundreds of ms regardless
421
- // of k.
412
+ // 3b. Corroborated-substitution bridge — refusal-path only (bridge.ts). The
413
+ // bridge's proposal source is the response's ONE top-k read — the same list
414
+ // recall already ranked above — never an exhaustive √N scan. The bridge's own
415
+ // candidate cap is 2·recallQueryK, so top-k proposals are exactly the budget
416
+ // it can consume, and every proposal is byte-verified downstream
417
+ // (exact-vs-approximate.md). Reuse the memoised `resonance()`; scanning every
418
+ // IVF cluster here once made every honest refusal cost hundreds of ms
419
+ // regardless of k.
422
420
  const wideIds = async () => (await pre.resonance()).map((h) => h.id);
423
421
 
424
422
  // Every gist-based tier has failed; before refusing, align the query
@@ -471,12 +469,12 @@ export async function recallByResonance(
471
469
  // prefixCompletion runs a few lines below and carries the three guards
472
470
  // this tier lacks — unreadable-continuation veto, sub-quantum
473
471
  // continuation, and UNIQUENESS (distinct continuations ⇒ refuse), which
474
- // is exactly what 4,300 competing values must trip. So this is not a
475
- // new rule and not a new threshold: it is deferring a prefix decision to
476
- // the tier that owns it (§2.5, one factored machinery). Byte-strict on
477
- // purpose — a candidate differing by case or punctuation ("what is the
478
- // capital of france" → "What is the capital of France?") is NOT a byte
479
- // prefix, keeps grounding here, and is unaffected.
472
+ // is exactly what 4,300 competing values must trip. So this is not a new
473
+ // rule and not a new threshold: it is deferring a prefix decision to the
474
+ // tier that owns it (match-project.md, one factored machinery).
475
+ // Byte-strict on purpose — a candidate differing by case or punctuation
476
+ // ("what is the capital of france" → "What is the capital of France?") is
477
+ // NOT a byte prefix, keeps grounding here, and is unaffected.
480
478
  const strictPrefix = g !== null &&
481
479
  cBytes.length > query.length &&
482
480
  indexOf(cBytes, query, 0) === 0;
@@ -542,15 +540,15 @@ export async function recallByResonance(
542
540
  }
543
541
  }
544
542
 
545
- // The refusal/echo decision. The echo returns a stored form's bytes AS
546
- // the answer — a near-identity claim about the query — and identity-grade
543
+ // The refusal/echo decision. The echo returns a stored form's bytes AS the
544
+ // answer — a near-identity claim about the query — and identity-grade
547
545
  // decisions are never made on an estimated score ("approximate scores may
548
- // rank and propose; they may never decide", §6.2): the RaBitQ estimate
549
- // overshooting the reach bar echoed a WRONG-entity neighbour ("capital of
550
- // Zamunda?" echoed the Armenia fact, observed). The bytes are read
551
- // anyway to be echoed, so the decision uses their EXACT fold: one river
552
- // fold of the top hit, measured in the same query-relative,
553
- // chance-corrected units as the tier above.
546
+ // rank and propose; they may never decide", exact-vs-approximate.md): the
547
+ // RaBitQ estimate overshooting the reach bar echoed a WRONG-entity neighbour
548
+ // ("capital of Zamunda?" echoed the Armenia fact, observed). The bytes are
549
+ // read anyway to be echoed, so the decision uses their EXACT fold: one river
550
+ // fold of the top hit, measured in the same query-relative, chance-corrected
551
+ // units as the tier above.
554
552
  const topBytes = read(ctx, top.id);
555
553
  const exact = topBytes.length > 0
556
554
  ? cosine(queryGist, gistOf(ctx, topBytes))
@@ -2,8 +2,8 @@
2
2
  // bytes (Grounding IV).
3
3
  //
4
4
  // This file is a CONFIGURATION of the shared frame reading in match.ts, not a
5
- // pipeline of its own. The three parts it configures live where §2.5 puts
6
- // them and are reachable by any mechanism:
5
+ // pipeline of its own. The three parts it configures live where
6
+ // match-project.md puts them and are reachable by any mechanism:
7
7
  //
8
8
  // matcher Precomputed.frames() — the frame INVENTORY: which ranked
9
9
  // candidates read as instances of the query's own frame, and
@@ -29,10 +29,10 @@
29
29
  // candidate's continuation UNSUBSTITUTED, so admitting a slot-gap there would
30
30
  // voice the corpus's filler for the asker's referent — the misreference
31
31
  // measured live on the trained store ("How do you say 'flurbish' in French?"
32
- // answered "the way to say hello is \"Bonjour\""). Nor is CAST rewired: its
32
+ // answered "the way to say hello is \"Bonjour\""). Nor is CAST rewired: its
33
33
  // frame gate is WEAVE-local while a slot is COHORT-local, and substituting one
34
- // population for the other is the error §2.7 names. The notion is made
35
- // AVAILABLE, never imposed.
34
+ // population for the other is the error commonality.md names. The notion is
35
+ // made AVAILABLE, never imposed.
36
36
 
37
37
  import type { MindContext } from "../types.js";
38
38
  import type { FrameInstance } from "../match.js";
@@ -52,16 +52,16 @@ import { rItem, rNode, traceFail } from "../trace.js";
52
52
  * agrees with nothing, so no carriage is attested — the same "two or no
53
53
  * constituent" reading frame-filler's contentRuns applies.
54
54
  *
55
- * THIS IS ALSO THE MECHANISM'S REACH. Evidence comes from the shared top-k
56
- * resonance, so a frame the corpus instantiates only ONCE within k is not
57
- * reachable here. Measured on the trained store: `How do you say 'flurbish'
58
- * in French?` finds one instance of its frame in the top 24 — the rest are
59
- * `How do you make …`, a different frame — so this abstains and recall's
60
- * scaffolding-dominated tier answers with the CORPUS's filler. That
61
- * misreference is recall's, and widening the supply is not the fix: the
62
- * exhaustive √N list recall's refusal path builds costs hundreds of
63
- * milliseconds and this runs before it. Abstaining on thin evidence is the
64
- * honest reading (§2.13). */
55
+ * THIS IS ALSO THE MECHANISM'S REACH. Evidence comes from the shared top-k
56
+ * resonance, so a frame the corpus instantiates only ONCE within k is not
57
+ * reachable here. Measured on the trained store: `How do you say 'flurbish' in
58
+ * French?` finds one instance of its frame in the top 24 — the rest are `How do
59
+ * you make …`, a different frame — so this abstains and recall's
60
+ * scaffolding-dominated tier answers with the CORPUS's filler. That
61
+ * misreference is recall's, and widening the supply is not the fix: the
62
+ * exhaustive √N list recall's refusal path builds costs hundreds of
63
+ * milliseconds and this runs before it. Abstaining on thin evidence is the
64
+ * honest reading (INVARIANTS.md). */
65
65
  const MIN_INSTANCES = 2;
66
66
 
67
67
  /** THE VOICING GATES — this mechanism's own reading of a pairing, applied here
@@ -135,7 +135,7 @@ function electFrame(
135
135
  let best: FrameInstance[] = [];
136
136
  for (const group of bySignature.values()) {
137
137
  // Ties keep the FIRST group in insertion order, which is resonance rank —
138
- // corpus-determined, like every other tie-break here (§2.1).
138
+ // corpus-determined, like every other tie-break here (determinism.md).
139
139
  if (group.length > best.length) best = group;
140
140
  }
141
141
  return best;
package/src/mind/mind.ts CHANGED
@@ -211,13 +211,12 @@ export interface MindOptions {
211
211
  host: import("../extension.js").ExtensionHost,
212
212
  ) => import("./pipeline-mechanism.js").PipelineMechanism)[];
213
213
  /** Measure the computational usage of every inference call — see
214
- * src/meter.ts. Off by default and free when off (one null check per
215
- * store read); on, each `respond`/`respondTurn` leaves a {@link
216
- * Mind.lastCost} report behind. Counters are deterministic, so two runs
217
- * of the same query on the same store are diffable; the millisecond
218
- * fields are not. Profiling NEVER changes an answer — but note that
219
- * attaching a RATIONALE does (traced responses bypass the ctx memos,
220
- * AGENTS §2.11), so profile without a trace. */
214
+ * src/meter.ts. Off by default and free when off (one null check per store
215
+ * read); on, each `respond`/`respondTurn` leaves a {@link Mind.lastCost}
216
+ * report behind. Counters are deterministic, so two runs of the same query on
217
+ * the same store are diffable; the millisecond fields are not. Profiling
218
+ * NEVER changes an answer — but attaching a RATIONALE does: a traced response
219
+ * bypasses the ctx memos (memoization.md), so profile without a trace. */
221
220
  profile?: boolean;
222
221
  /** Content canonicalizer applied to EVERY response (any modality) for
223
222
  * equivalence-class resolution — see src/canon.ts. Text entry points
@@ -5,7 +5,7 @@
5
5
  // a list of PipelineMechanism objects — it never imports a mechanism-specific
6
6
  // type and never has a special-case branch for any mechanism.
7
7
  //
8
- // The four constraints of the free-will architecture (§14.5):
8
+ // The four constraints of the free-will architecture (mechanism-market.md):
9
9
  // 1. DECOUPLING — mechanisms import nothing from each other or from pipeline.
10
10
  // 2. DECLARED COMPETENCE — floor() returns null when impossible, a number when
11
11
  // possible. Binary, auditable, no learned scores.
@@ -148,17 +148,17 @@ export class Precomputed {
148
148
  );
149
149
  }
150
150
 
151
- // REMOVED — the WIDE exhaustive-√N resonance list (`wideResonance`). It ran
151
+ // REMOVED — the WIDE exhaustive-√N resonance list (`wideResonance`). It ran
152
152
  // `resonate(guide, √N, exhaustive=true)` whenever the top hit cleared
153
- // conceptThreshold, so consumers could look "past the top-k". Every consumer
153
+ // conceptThreshold, so consumers could look "past the top-k". Every consumer
154
154
  // only ever needed ≤ 2·recallQueryK proposals (the substitution bridge's own
155
155
  // candidate cap) or a content-addressed answer (prefix completion's
156
- // formsOpenedBy), and every proposal is byte-verified downstream (§2.3), so
157
- // the exhaustive scan bought recall at O(index) cost for an O(k) need —
158
- // measured: 244K annVectorReads per refusing query, ~1.5 s, every answer
159
- // byte-identical to a top-k read. The two consumers now read `resonance()`
160
- // (the one top-k read) and the write side's window index respectively — see
161
- // recall.ts and prefix-completion.ts.
156
+ // formsOpenedBy), and every proposal is byte-verified downstream
157
+ // (exact-vs-approximate.md), so the exhaustive scan bought recall at O(index)
158
+ // cost for an O(k) need — measured: 244K annVectorReads per refusing query,
159
+ // ~1.5 s, every answer byte-identical to a top-k read. The two consumers now
160
+ // read `resonance()` (the one top-k read) and the write side's window index
161
+ // respectively — see recall.ts and prefix-completion.ts.
162
162
 
163
163
  private _frames?: Promise<ReadonlyArray<FrameInstance>>;
164
164
  /** THE FRAME INVENTORY — every ranked candidate that reads as an instance of
@@ -166,14 +166,16 @@ export class Precomputed {
166
166
  * ({@link FrameInstance}). The one place the engine represents "a position
167
167
  * whose occupant comes from the context rather than the corpus".
168
168
  *
169
- * AN INVENTORY, NOT AN ELECTION. It reports every pairing and elects no
170
- * frame, deliberately: a slot is a property of a PAIRING, not of the query,
171
- * and different candidates put slots in different places. Committing to one
172
- * reading here would push whichever consumer asked first onto everyone else
173
- * — the market's decoupling (§2.6) broken from inside the shared container,
174
- * and the population error §2.7 names. Each consumer groups and commits
175
- * for its own question; reference elects the modal slot signature, and a
176
- * consumer wanting a different reading is not fighting this one.
169
+ * AN INVENTORY, NOT AN ELECTION. It reports every pairing and elects no
170
+ * frame,
171
+ * deliberately: a slot is a property of a PAIRING, not of the query, and
172
+ * different candidates put slots in different places. Committing to one
173
+ * reading here would push whichever consumer asked first onto everyone else —
174
+ * the market's decoupling (mechanism-market.md) broken from inside the shared
175
+ * container, and the population error commonality.md names. Each consumer
176
+ * groups and commits for its own question; reference elects the modal slot
177
+ * signature, and a consumer wanting a different reading is not fighting this
178
+ * one.
177
179
  *
178
180
  * NO LICENCE EITHER. Knowing a span is variable is safe for every consumer
179
181
  * — it can only improve an alignment. Knowing one may be VOICED through is
@@ -189,9 +191,10 @@ export class Precomputed {
189
191
  const capBytes = this.query.length * W;
190
192
  const out: FrameInstance[] = [];
191
193
  for (const h of await this.resonance()) {
192
- // REJECT BY LENGTH BEFORE RECONSTRUCTING (§2.8): `contentLen` is an
193
- // indexed read, `bytesPrefix` rebuilds a subtree. ONLY the phrase-scale
194
- // cap is applied — it is a bounded-read discipline, not a judgement.
194
+ // REJECT BY LENGTH BEFORE RECONSTRUCTING (bounded-reads.md):
195
+ // `contentLen` is an indexed read, `bytesPrefix` rebuilds a subtree.
196
+ // ONLY the phrase-scale cap is applied — it is a bounded-read
197
+ // discipline, not a judgement.
195
198
  //
196
199
  // A LOWER bound was here too (`dominates(len, query.length)`, on the
197
200
  // reasoning that a candidate shorter than half the query cannot supply
@@ -519,7 +522,8 @@ function computeWeave(
519
522
  // IDF — gates the aligner has no equivalent of.
520
523
  //
521
524
  // So the climb PROPOSES the pairing (which structure, which query span) and
522
- // bytes DECIDE its terms (§2.3). Three gates, each one measured:
525
+ // bytes DECIDE its terms (exact-vs-approximate.md). Three gates, each one
526
+ // measured:
523
527
  //
524
528
  // • it may only take query bytes NO literal run claimed. Run inline with
525
529
  // phase 1 this did the opposite of "exact decides" — a higher-ranked
@@ -130,15 +130,15 @@ export interface NarrowDecisionData {
130
130
  }
131
131
 
132
132
  /** Structured payload of the "regimePrediction" rationale step — the R8
133
- * observation exposed as data. After the first mechanism (cover, which §2.6
134
- * runs first) grounds or abstains, the market's whole outcome is already
135
- * determined by the one cost ladder: the consensus climb runs exactly when
136
- * `worthRunning(2 * STEP)` is true — CAST (floor 2·STEP) is the cheapest
137
- * mechanism that first-touches it, and confluence (3·STEP) / extraction
138
- * (CONCEPT+STEP) are only reached after CAST is. An incumbent at or below
139
- * that floor prunes CAST and, with it, the climb (retrieval); anything above
140
- * — or no incumbent — runs the full market and the climb (composition).
141
- * Purely observational; never read by inference. */
133
+ * observation exposed as data. After the first mechanism (cover, which
134
+ * mechanism-market.md runs first) grounds or abstains, the market's whole
135
+ * outcome is already determined by the one cost ladder: the consensus climb
136
+ * runs exactly when `worthRunning(2 * STEP)` is true — CAST (floor 2·STEP) is
137
+ * the cheapest mechanism that first-touches it, and confluence (3·STEP) /
138
+ * extraction (CONCEPT+STEP) are only reached after CAST is. An incumbent at or
139
+ * below that floor prunes CAST and, with it, the climb (retrieval); anything
140
+ * above — or no incumbent — runs the full market and the climb (composition).
141
+ * Purely observational; never read by inference. */
142
142
  export interface RegimePredictionData {
143
143
  version: 1;
144
144
  /** retrieval | composition — the two regimes R1 measured as a ~100× cost
@@ -187,13 +187,13 @@ export async function think(
187
187
  // ── Pre-computation ──────────────────────────────────────────────────
188
188
  const mechanisms = mechs ?? defaultMechanisms;
189
189
  const meter = ctx.meter;
190
- // recognition is a shared analysis (§2.14 contract 5): it does the query's
190
+ // recognition is a shared analysis (meter.md contract 5): it does the query's
191
191
  // own store work (perceive → foldTree → resolve), which used to land in
192
192
  // `think` and in nothing narrower — the meter's one accounting surface must
193
193
  // charge it to itself, exactly as attention/weave/resonance are charged.
194
- // SYNCHRONOUS phase: recognition is on the sync side of §2.10's seam, so it
195
- // is timed with `timeSync` — wrapping it in a promise would make a profiled
196
- // response await where an unprofiled one does not.
194
+ // SYNCHRONOUS phase: recognition is on the sync side of meter.md's seam, so
195
+ // it is timed with `timeSync` — wrapping it in a promise would make a
196
+ // profiled response await where an unprofiled one does not.
197
197
  const rec = meter
198
198
  ? meter.timeSync("recognise", () => recognise(ctx, query))
199
199
  : recognise(ctx, query);
@@ -225,15 +225,15 @@ export async function think(
225
225
  }
226
226
  }
227
227
 
228
- // Phase 2: the shared pre-computation container. Eager fields only
229
- // (recognition, computed spans, guide) — every expensive analysis
230
- // (consensus climb, weave, span-shape classification) is a lazily-cached
231
- // method on Precomputed, first-touched by whichever mechanism's floor
232
- // survives its cheap gates and the worthRunning check. A query no
233
- // mechanism climbs for (e.g. one an extension decided) never climbs.
234
- // NOT phased: the constructor itself is trivial (it only derives `k`), so a
235
- // phase here would add a zero-work entry to every profiled report — the meter
236
- // attributes WORK (§2.14); the trace already represents structure.
228
+ // Phase 2: the shared pre-computation container. Eager fields only
229
+ // (recognition, computed spans, guide) — every expensive analysis (consensus
230
+ // climb, weave, span-shape classification) is a lazily-cached method on
231
+ // Precomputed, first-touched by whichever mechanism's floor survives its
232
+ // cheap gates and the worthRunning check. A query no mechanism climbs for
233
+ // (e.g. one an extension decided) never climbs. NOT phased: the constructor
234
+ // itself is trivial (it only derives `k`), so a phase here would add a
235
+ // zero-work entry to every profiled report — the meter attributes WORK
236
+ // (meter.md); the trace already represents structure.
237
237
  const pre = new Precomputed(ctx, query, rec, computed, ctx._edgeGuide);
238
238
 
239
239
  // ── Grounding: ONE lightest-derivation choice among the mechanisms ────
@@ -298,16 +298,17 @@ export async function think(
298
298
  const worthRunning = (floor: number) =>
299
299
  best === null || grade(floor) < grade(best.weight);
300
300
 
301
- // REGIME PREDICTION (R8) — observational only. Once the FIRST mechanism has
302
- // had its turn (cover, which §2.6 places first and floors at 0), the market's
303
- // outcome is already determined by the one cost ladder: the consensus climb
304
- // runs exactly when `worthRunning(2 * STEP)` is true — CAST (floor 2·STEP) is
305
- // the cheapest mechanism that first-touches it, so an incumbent at or below
306
- // grade 2 prunes CAST and, with it, confluence (3·STEP) and extraction
307
- // (CONCEPT+STEP) (retrieval); anything above — or no incumbent — runs the
308
- // full market and the climb (composition). The predicate is `worthRunning`,
309
- // the same function the loop itself uses — nothing is computed here that the
310
- // engine had not already computed, and nothing is read back by inference.
301
+ // REGIME PREDICTION (R8) — observational only. Once the FIRST mechanism has
302
+ // had its turn (cover, which mechanism-market.md places first and floors at
303
+ // 0), the market's outcome is already determined by the one cost ladder: the
304
+ // consensus climb runs exactly when `worthRunning(2 * STEP)` is true — CAST
305
+ // (floor 2·STEP) is the cheapest mechanism that first-touches it, so an
306
+ // incumbent at or below grade 2 prunes CAST and, with it, confluence (3·STEP)
307
+ // and extraction (CONCEPT+STEP) (retrieval); anything above — or no incumbent
308
+ // — runs the full market and the climb (composition). The predicate is
309
+ // `worthRunning`, the same function the loop itself uses — nothing is
310
+ // computed here that the engine had not already computed, and nothing is read
311
+ // back by inference.
311
312
  //
312
313
  // EMITTED BEFORE THE SECOND MECHANISM'S FLOOR, never after some mechanism's
313
314
  // run: a "prediction" published after the fact could assert "the climb will
@@ -59,11 +59,11 @@ export function perceiveKey(
59
59
  /** Perceive input into a content-defined tree (the river fold).
60
60
  * Deterministic — identical bytes always produce an identical tree.
61
61
  *
62
- * `boundaries` is an optional sorted list of proper byte offsets where the
63
- * fold must split so that each prefix segment folds identically to how it
64
- * folded when it was learned (§10.3 stable-prefix contract). Only the
65
- * CALLER — who assembled the multi-turn context — knows where those
66
- * boundaries are; the geometry never guesses them from the bytes. */
62
+ * `boundaries` is an optional sorted list of proper byte offsets where the fold
63
+ * must split so that each prefix segment folds identically to how it folded
64
+ * when it was learned (fold-contract.md stable-prefix contract). Only the
65
+ * CALLER — who assembled the multi-turn context — knows where those boundaries
66
+ * are; the geometry never guesses them from the bytes. */
67
67
  export function perceive(
68
68
  ctx: MindContext,
69
69
  input: Input,
@@ -34,19 +34,20 @@ import type { Leaf, Site } from "./graph-search.js";
34
34
  *
35
35
  * Both O(n · maxGroup) bounded O(1) probes — never a scan of the corpus.
36
36
  *
37
- * ONE READING PER BYTE STREAM, deliberately: there is no "cheap mode" that
38
- * skips the edge-trim fallbacks. A `trimmed` variant was tried and REFUTED
39
- * twice over. Its premise — "the trims only recover misaligned FRAGMENTS, so
40
- * a consumer whose gate rejects fragments loses nothing" — is false: the
41
- * left/right trim loops below exist precisely to find WHOLE trained forms
42
- * embedded at an offset the query's own fold did not cut, and such a form has
43
- * no structural parents or containers, so it passes the pivot's fragment gate
44
- * and is exactly the candidate a multi-hop chain steps through. Skipping them
45
- * narrows the pivot's evidence silently. And a per-caller variant has to key
46
- * the memo by the variant, which breaks the "computed at most once" property
47
- * (§2.11): the pipeline recognises a grounded answer untrimmed for
48
- * `preConsumed`, and the pivot then recognises the same bytes again — the
49
- * saving inverts into a doubling on the path it was measured for. */
37
+ * ONE READING PER BYTE STREAM, deliberately: there is no "cheap mode" that
38
+ * skips
39
+ * the edge-trim fallbacks. A `trimmed` variant was tried and REFUTED twice
40
+ * over. Its premise — "the trims only recover misaligned FRAGMENTS, so a
41
+ * consumer whose gate rejects fragments loses nothing" — is false: the
42
+ * left/right trim loops below exist precisely to find WHOLE trained forms
43
+ * embedded at an offset the query's own fold did not cut, and such a form has
44
+ * no structural parents or containers, so it passes the pivot's fragment gate
45
+ * and is exactly the candidate a multi-hop chain steps through. Skipping them
46
+ * narrows the pivot's evidence silently. And a per-caller variant has to key
47
+ * the memo by the variant, which breaks the "computed at most once" property
48
+ * (memoization.md): the pipeline recognises a grounded answer untrimmed for
49
+ * `preConsumed`, and the pivot then recognises the same bytes again — the
50
+ * saving inverts into a doubling on the path it was measured for. */
50
51
  export function recognise(
51
52
  ctx: MindContext,
52
53
  bytes: Uint8Array,
@@ -180,16 +181,15 @@ function recogniseImpl(ctx: MindContext, bytes: Uint8Array): Recognition {
180
181
  // and read the answer from `starts`, which is exactly {0, W, 2W, …}
181
182
  // because riverFold groups fixed-arity — arithmetic, not evidence.
182
183
  //
183
- // Measured on the 17.9M-node store, over the sites of 7 probes (1 good,
184
- // 11 junk by hand-labelling, corrected for whole-query forms):
185
- // len >= W rejects "hi"(2) "of"(2) "is"(2) "di"(2) "the"(3),
186
- // admits "Eiffel Tower"(12) and both whole-query forms
187
- // len >= W-1 admits "the" — W-1 is the write side's straddle
188
- // neighbour for RETRIEVAL, never a claim about units
189
- // §2.7 saturation admits 11/11 junk: edgeAncestors on a site node
190
- // reaches 1..48 contexts, so dominates(ctx, N) needs
191
- // ctx > 162805 and never fires; every site reads DISC
192
- // rarity does not separate: "hi" has 1 container, "the" 572
184
+ // Measured on the 17.9M-node store, over the sites of 7 probes (1 good, 11
185
+ // junk by hand-labelling, corrected for whole-query forms): len >= W
186
+ // rejects "hi"(2) "of"(2) "is"(2) "di"(2) "the"(3), admits "Eiffel
187
+ // Tower"(12) and both whole-query forms len >= W-1 admits "the" — W-1 is
188
+ // the write side's straddle neighbour for RETRIEVAL, never a claim about
189
+ // units commonality.md saturation admits 11/11 junk: edgeAncestors on a
190
+ // site node reaches 1..48 contexts, so dominates(ctx, N) needs ctx > 162805
191
+ // and never fires; every site reads DISC rarity does not separate: "hi" has
192
+ // 1 container, "the" 572
193
193
  //
194
194
  // A span covering the WHOLE query is exempt: then it is not a fragment of
195
195
  // something longer, it is the question ("hi" asked on its own).
@@ -517,7 +517,7 @@ function recogniseImpl(ctx: MindContext, bytes: Uint8Array): Recognition {
517
517
  start: number,
518
518
  end: number,
519
519
  canonBudget: boolean,
520
- ): void => {
520
+ ): boolean => {
521
521
  // Any span at least one river window wide is worth a probe. This used
522
522
  // to stop at `chainReach(W)` — "the chain already covers anything that
523
523
  // short" — and that premise does not hold for every embedded form: the
@@ -531,14 +531,17 @@ function recogniseImpl(ctx: MindContext, bytes: Uint8Array): Recognition {
531
531
  // starts on a fold cut nor ends on a node edge was unreachable by either
532
532
  // tier — the exact site whose loss `tryChain`'s own note records as "the
533
533
  // pivot dies with the site and multi-hop goes silent". The interior
534
- // pass below spends the same budget on those pairs.
535
- if (end - start < W) return;
534
+ // pass below spends the same budget on those pairs. Returns whether it
535
+ // emitted, so a caller can retry a trimmed edge on the miss path only.
536
+ if (end - start < W) return false;
536
537
  if (flatProbe(start, end) === null) {
537
- if (!canonBudget) return;
538
- if (!canonAdmits(start, end)) return;
538
+ if (!canonBudget) return false;
539
+ if (!canonAdmits(start, end)) return false;
539
540
  }
540
541
  const id = resolveSpan(start, end);
541
- if (id !== null) emit(start, end, id);
542
+ if (id === null) return false;
543
+ emit(start, end, id);
544
+ return true;
542
545
  };
543
546
  // A CUMULATIVE BYTE BUDGET, SPENT SHORTEST-SPAN-FIRST.
544
547
  //
@@ -574,10 +577,7 @@ function recogniseImpl(ctx: MindContext, bytes: Uint8Array): Recognition {
574
577
  const span = end - start;
575
578
  const afford = span <= budget;
576
579
  if (afford) budget -= span;
577
- probe(start, end, afford);
578
- // Always keep walking: the exact route is unbudgeted, so running out
579
- // of canon budget must not stop the scan.
580
- return true;
580
+ return probe(start, end, afford);
581
581
  };
582
582
  const prefixes = ordered.filter((e) => e > 0).sort((a, b) => a - b);
583
583
  const suffixes = ordered
@@ -585,9 +585,24 @@ function recogniseImpl(ctx: MindContext, bytes: Uint8Array): Recognition {
585
585
  .sort((a, b) => b - a);
586
586
  for (let i = 0; i < Math.max(prefixes.length, suffixes.length); i++) {
587
587
  // Interleaved so neither edge starves the other when the budget runs
588
- // out — a query can carry a trained form at either end.
589
- if (i < prefixes.length && !spend(0, prefixes[i])) break;
590
- if (i < suffixes.length && !spend(suffixes[i], bytes.length)) break;
588
+ // out — a query can carry a trained form at either end. The scan never
589
+ // stops on a miss: the exact route is unbudgeted, so running out of
590
+ // canon budget must not stop it.
591
+ if (i < prefixes.length) spend(0, prefixes[i]);
592
+ if (i < suffixes.length) {
593
+ const s = suffixes[i];
594
+ // An edge form can end ONE byte before the edge does: a query often
595
+ // closes with a separator ("…is Timur Bekmambetov.") that belongs
596
+ // BETWEEN forms, and the text canonicalizer passes punctuation through,
597
+ // so the full-edge probe can never match it. Retry the trimmed edge
598
+ // ON THE MISS PATH ONLY — self-verifying (resolve decides, so a wrong
599
+ // trim can never emit), the same ±1 edge-trim discipline the canon-miss
600
+ // fallback above already trusts, and the hit path pays nothing. A form
601
+ // LONGER than `chainReach` at the sentence end (measured: "Timur
602
+ // Bekmambetov", 17 bytes) is recovered only here — the interior pass is
603
+ // capped at `chainReach`.
604
+ if (!spend(s, bytes.length)) spend(s, bytes.length - 1);
605
+ }
591
606
  }
592
607
  // INTERIOR pairs within the same `chainReach(W)` span bound the chain
593
608
  // trusts — the dead zone the gate above used to leave: a form that neither
@@ -380,15 +380,15 @@ export async function pivotInto(
380
380
  // Byte containment, longest wins — the answer literally contains the
381
381
  // pivot's bytes, and the biggest well-evidenced span is the real pivot.
382
382
  //
383
- // REAL SATURATION, not a hard cap: the score IS the candidate's byte
384
- // length, so the scan is DECIDED the moment the first candidate that passes
385
- // every filter is found in DESCENDING length order — a shorter candidate can
386
- // never outscore it. `contentLen` (the prefix-capped length read, §2.8) is
387
- // the cheap ordering key, and the first-inserted tie-break is made explicit
388
- // (`a.index - b.index`) so equal lengths keep `scored`'s insertion order —
389
- // exactly the tie argmaxBy(strict) used to keep. The bytes of at most ONE
390
- // winning candidate are read; every shorter candidate the probes proposed is
391
- // skipped without reconstruction, where the old argmax read them all.
383
+ // REAL SATURATION, not a hard cap: the score IS the candidate's byte length,
384
+ // so the scan is DECIDED the moment the first candidate that passes every
385
+ // filter is found in DESCENDING length order — a shorter candidate can never
386
+ // outscore it. `contentLen` (the prefix-capped length read, bounded-reads.md)
387
+ // is the cheap ordering key, and the first-inserted tie-break is made
388
+ // explicit (`a.index - b.index`) so equal lengths keep `scored`'s insertion
389
+ // order — exactly the tie argmaxBy(strict) used to keep. The bytes of at most
390
+ // ONE winning candidate are read; every shorter candidate the probes proposed
391
+ // is skipped without reconstruction, where the old argmax read them all.
392
392
  const ranked = [...scored.keys()]
393
393
  .map((id, index) => ({
394
394
  id,
@@ -399,11 +399,11 @@ export async function pivotInto(
399
399
  let pivotId: number | null = null;
400
400
  for (const c of ranked) {
401
401
  const id = c.id;
402
- // A ZERO-LENGTH candidate is not a pivot. `argmaxBy(…, 0, strict)` used to
402
+ // A ZERO-LENGTH candidate is not a pivot. `argmaxBy(…, 0, strict)` used to
403
403
  // carry this floor in its threshold argument, and dropping it here would
404
404
  // admit an empty node: `indexOf(answer, <empty>)` returns 0, so every
405
- // filter below passes and the chain would hop through nothing (§2.13 —
406
- // empty bytes are truthy).
405
+ // filter below passes and the chain would hop through nothing
406
+ // (INVARIANTS.md — empty bytes are truthy).
407
407
  if (c.len === 0) continue;
408
408
  // A PIVOT MUST BE A THING THE CORPUS DEPOSITED, NOT A PIECE OF ONE.
409
409
  // "Longest wins" ranks candidates but never asks whether the winner is
@@ -439,15 +439,15 @@ export async function pivotInto(
439
439
  // a span that was never a fact on its own is not one to step through.
440
440
  // No constant enters — it is a structural predicate, not a threshold.
441
441
  if (ctx.store.hasParents(id) || ctx.store.hasContainers(id)) continue;
442
- // A candidate whose bytes are LONGER than the answer cannot be a
443
- // substring of it — `indexOf` would return −1 regardless. Prune by
444
- // length BEFORE reconstructing the bytes: `read` is an UNCAPPED read
445
- // (AGENTS §2.8), and a resonated context far longer than the answer is
446
- // exactly the candidate that makes it cost a whole deposit's worth of
447
- // reconstruction for a containment test that must fail. `contentLen`
448
- // with the `answer.length + 1` cap is the prefix-capped length read the
449
- // same contract prescribes; the prune is byte-identical to the old
450
- // `indexOf` miss (it returns −1 for a needle longer than the haystack).
442
+ // A candidate whose bytes are LONGER than the answer cannot be a substring
443
+ // of it — `indexOf` would return −1 regardless. Prune by length BEFORE
444
+ // reconstructing the bytes: `read` is an UNCAPPED read (bounded-reads.md),
445
+ // and a resonated context far longer than the answer is exactly the
446
+ // candidate that makes it cost a whole deposit's worth of reconstruction
447
+ // for a containment test that must fail. `contentLen` with the
448
+ // `answer.length + 1` cap is the prefix-capped length read the same
449
+ // contract prescribes; the prune is byte-identical to the old `indexOf`
450
+ // miss (it returns −1 for a needle longer than the haystack).
451
451
  if (c.len > answer.length) continue;
452
452
  const bytes = read(ctx, id);
453
453
  if (indexOf(answer, bytes, 0) < 0) continue;