@hviana/sema 0.8.0 → 0.8.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (119) hide show
  1. package/AGENTS.md +22 -1
  2. package/DATASETS.md +1 -1
  3. package/dist/example/train_base/config.js +2 -2
  4. package/dist/example/train_base/corpora/massive.js +1 -1
  5. package/dist/example/train_base/readers.js +1 -1
  6. package/dist/src/config.d.ts +17 -0
  7. package/dist/src/config.js +18 -0
  8. package/dist/src/geometry.d.ts +10 -10
  9. package/dist/src/geometry.js +25 -24
  10. package/dist/src/meter.d.ts +29 -12
  11. package/dist/src/meter.js +58 -14
  12. package/dist/src/mind/attention.js +12 -12
  13. package/dist/src/mind/bridge.d.ts +8 -8
  14. package/dist/src/mind/bridge.js +33 -32
  15. package/dist/src/mind/corpus.d.ts +40 -0
  16. package/dist/src/mind/corpus.js +149 -0
  17. package/dist/src/mind/graph-search.d.ts +7 -8
  18. package/dist/src/mind/graph-search.js +244 -32
  19. package/dist/src/mind/index.d.ts +3 -1
  20. package/dist/src/mind/index.js +1 -0
  21. package/dist/src/mind/junction.d.ts +1 -1
  22. package/dist/src/mind/junction.js +8 -8
  23. package/dist/src/mind/learning.js +36 -35
  24. package/dist/src/mind/match.d.ts +8 -3
  25. package/dist/src/mind/match.js +156 -71
  26. package/dist/src/mind/mechanisms/cast.js +18 -2
  27. package/dist/src/mind/mechanisms/cover.js +19 -12
  28. package/dist/src/mind/mechanisms/prefix-completion.js +24 -24
  29. package/dist/src/mind/mechanisms/recall.js +38 -40
  30. package/dist/src/mind/mechanisms/reference.js +16 -16
  31. package/dist/src/mind/mind.d.ts +61 -7
  32. package/dist/src/mind/mind.js +72 -2
  33. package/dist/src/mind/pipeline-mechanism.d.ts +10 -8
  34. package/dist/src/mind/pipeline-mechanism.js +25 -21
  35. package/dist/src/mind/pipeline.d.ts +9 -9
  36. package/dist/src/mind/pipeline.js +49 -29
  37. package/dist/src/mind/primitives.d.ts +5 -5
  38. package/dist/src/mind/primitives.js +5 -5
  39. package/dist/src/mind/reasoning.d.ts +5 -1
  40. package/dist/src/mind/reasoning.js +54 -1
  41. package/dist/src/mind/recognition.d.ts +14 -13
  42. package/dist/src/mind/recognition.js +23 -23
  43. package/dist/src/mind/resonance.js +21 -21
  44. package/dist/src/mind/traverse.d.ts +54 -52
  45. package/dist/src/mind/traverse.js +83 -73
  46. package/dist/src/mind/types.d.ts +26 -4
  47. package/dist/src/store.d.ts +12 -12
  48. package/dist/src/store.js +12 -12
  49. package/docs/INDEX.md +2 -2
  50. package/docs/architecture/exact-vs-approximate.md +2 -1
  51. package/docs/architecture/fold-contract.md +1 -1
  52. package/docs/failures/tempting-but-wrong.md +33 -5
  53. package/docs/harness/gates.md +7 -7
  54. package/example/train_base/config.ts +2 -2
  55. package/example/train_base/corpora/massive.ts +1 -1
  56. package/example/train_base/readers.ts +1 -1
  57. package/jsr.json +1 -1
  58. package/package.json +1 -1
  59. package/src/config.ts +35 -0
  60. package/src/geometry.ts +25 -24
  61. package/src/meter.ts +61 -14
  62. package/src/mind/attention.ts +12 -12
  63. package/src/mind/bridge.ts +33 -32
  64. package/src/mind/corpus.ts +202 -0
  65. package/src/mind/graph-search.ts +261 -31
  66. package/src/mind/index.ts +8 -1
  67. package/src/mind/junction.ts +8 -8
  68. package/src/mind/learning.ts +36 -35
  69. package/src/mind/match.ts +163 -73
  70. package/src/mind/mechanisms/cast.ts +17 -1
  71. package/src/mind/mechanisms/cover.ts +18 -12
  72. package/src/mind/mechanisms/prefix-completion.ts +24 -24
  73. package/src/mind/mechanisms/recall.ts +38 -40
  74. package/src/mind/mechanisms/reference.ts +16 -16
  75. package/src/mind/mind.ts +129 -7
  76. package/src/mind/pipeline-mechanism.ts +25 -21
  77. package/src/mind/pipeline.ts +63 -38
  78. package/src/mind/primitives.ts +5 -5
  79. package/src/mind/reasoning.ts +55 -0
  80. package/src/mind/recognition.ts +23 -23
  81. package/src/mind/resonance.ts +21 -21
  82. package/src/mind/traverse.ts +83 -73
  83. package/src/mind/types.ts +30 -4
  84. package/src/store.ts +20 -20
  85. package/test/08-storage.test.mjs +1 -1
  86. package/test/100-complete-grounding-trace.test.mjs +109 -0
  87. package/test/101-alignment-gap-bound.test.mjs +106 -0
  88. package/test/102-production-composes-at-scale.test.mjs +110 -0
  89. package/test/103-alignment-gap-budget.test.mjs +89 -0
  90. package/test/104-composition-is-reported.test.mjs +90 -0
  91. package/test/105-derive-through-reports-its-refusal.test.mjs +113 -0
  92. package/test/106-the-join-fires.test.mjs +94 -0
  93. package/test/107-the-join-is-counted.test.mjs +81 -0
  94. package/test/108-the-join-chains.test.mjs +78 -0
  95. package/test/109-the-pivot-is-counted.test.mjs +60 -0
  96. package/test/110-the-reasoner-stops-when-the-question-is-answered.test.mjs +91 -0
  97. package/test/111-the-cover-assembly-is-counted.test.mjs +74 -0
  98. package/test/112-the-exploration-does-not-grow-with-the-hub.test.mjs +89 -0
  99. package/test/113-the-rationale-payload-is-bounded.test.mjs +84 -0
  100. package/test/114-alignment-budget-is-per-sweep.test.mjs +93 -0
  101. package/test/116-the-extension-is-gated-by-the-pipelines-own-remainder.test.mjs +100 -0
  102. package/test/117-corpus-search.test.mjs +171 -0
  103. package/test/14-scaling.test.mjs +10 -7
  104. package/test/35-prefix-edge.test.mjs +1 -1
  105. package/test/40-choosenext-scale-guard.test.mjs +16 -17
  106. package/test/56-bridge-identity-admission.test.mjs +6 -6
  107. package/test/70-prefix-completion.test.mjs +4 -3
  108. package/test/72-prefix-candidate-supply.test.mjs +3 -3
  109. package/test/73-scaffolding-only-bridge-abstains.test.mjs +6 -6
  110. package/test/75-multiturn-context-optimisation.test.mjs +5 -5
  111. package/test/76-reference-binding.test.mjs +6 -1
  112. package/test/84-composed-answer-honesty.test.mjs +5 -6
  113. package/test/88-dependency-footprint.test.mjs +1 -1
  114. package/test/89-completion-recursion.test.mjs +47 -19
  115. package/test/90-connector-read-cap.test.mjs +10 -8
  116. package/test/93-regime-prediction.test.mjs +10 -10
  117. package/test/94-cross-region-budget.test.mjs +2 -2
  118. package/test/95-wide-resonance-removed.test.mjs +8 -7
  119. package/test/96-bytes-walk-termination.test.mjs +3 -3
@@ -159,24 +159,22 @@ export async function recallByResonance(ctx, query, pre) {
159
159
  }
160
160
  }
161
161
  }
162
- // The query-relative grounding fraction, shared by tiers 2–4 — gated on
163
- // the FRACTION OF THE QUERY the grounding explains, not the raw cosine.
164
- // Root gists are unit vectors, but their magnitudes are recoverable from
165
- // the byte lengths (‖·‖ = √len under the linear fold):
166
- // cos = shared/√(lenQ·lenG), so shared/lenQ = cos·√(lenG/lenQ).
167
- // The raw cosine punished honest containment — a query fully inside a
168
- // longer grounded answer scored √(lenQ/lenG) and was refused — and let a
169
- // long answer sharing only scaffolding pass; the query-relative fraction
170
- // measures exactly what the reach bar means: how much of THE QUERY the
171
- // store accounts for.
172
- // Chance similarity survives the length conversion AMPLIFIED: the same
173
- // √(lenG/lenQ) factor that converts an honest shared fraction into a
174
- // query-relative one multiplies the estimator/chance floor too, so a long
175
- // stored form (√(lenG/lenQ) ≈ 10 at 100×) lifted a noise-level cosine past
176
- // the reach bar and grounded pure gibberish (observed). Only the
177
- // ABOVE-CHANCE part of the similarity is evidence of shared content —
178
- // subtract the significance bar (3/√D, §8.3) before converting. Derived
179
- // from the existing bars; never tuned.
162
+ // The query-relative grounding fraction, shared by tiers 2–4 — gated on the
163
+ // FRACTION OF THE QUERY the grounding explains, not the raw cosine. Root
164
+ // gists are unit vectors, but their magnitudes are recoverable from the byte
165
+ // lengths (‖·‖ = √len under the linear fold): cos = shared/√(lenQ·lenG), so
166
+ // shared/lenQ = cos·√(lenG/lenQ). The raw cosine punished honest containment
167
+ // — a query fully inside a longer grounded answer scored √(lenQ/lenG) and was
168
+ // refused — and let a long answer sharing only scaffolding pass; the
169
+ // query-relative fraction measures exactly what the reach bar means: how much
170
+ // of THE QUERY the store accounts for. Chance similarity survives the length
171
+ // conversion AMPLIFIED: the same √(lenG/lenQ) factor that converts an honest
172
+ // shared fraction into a query-relative one multiplies the estimator/chance
173
+ // floor too, so a long stored form (√(lenG/lenQ) ≈ 10 at 100×) lifted a
174
+ // noise-level cosine past the reach bar and grounded pure gibberish
175
+ // (observed). Only the ABOVE-CHANCE part of the similarity is evidence of
176
+ // shared content — subtract the significance bar (3/√D, thresholds.md) before
177
+ // converting. Derived from the existing bars; never tuned.
180
178
  const sig = significanceBar(ctx.store.D);
181
179
  const reach = reachThreshold(ctx.space.maxGroup);
182
180
  const fracOfQuery = (cos, otherLen) => Math.min(1, Math.max(0, cos - sig) *
@@ -305,14 +303,14 @@ export async function recallByResonance(ctx, query, pre) {
305
303
  }
306
304
  }
307
305
  }
308
- // 3b. Corroborated-substitution bridge — refusal-path only (bridge.ts).
309
- // The bridge's proposal source is the response's ONE top-k read — the same
310
- // list recall already ranked above — never an exhaustive √N scan. The
311
- // bridge's own candidate cap is 2·recallQueryK, so top-k proposals are
312
- // exactly the budget it can consume, and every proposal is byte-verified
313
- // downstream (§2.3). Reuse the memoised `resonance()`; scanning every IVF
314
- // cluster here once made every honest refusal cost hundreds of ms regardless
315
- // of k.
306
+ // 3b. Corroborated-substitution bridge — refusal-path only (bridge.ts). The
307
+ // bridge's proposal source is the response's ONE top-k read — the same list
308
+ // recall already ranked above — never an exhaustive √N scan. The bridge's own
309
+ // candidate cap is 2·recallQueryK, so top-k proposals are exactly the budget
310
+ // it can consume, and every proposal is byte-verified downstream
311
+ // (exact-vs-approximate.md). Reuse the memoised `resonance()`; scanning every
312
+ // IVF cluster here once made every honest refusal cost hundreds of ms
313
+ // regardless of k.
316
314
  const wideIds = async () => (await pre.resonance()).map((h) => h.id);
317
315
  // Every gist-based tier has failed; before refusing, align the query
318
316
  // byte-for-byte against the trained contexts its own stored windows
@@ -362,12 +360,12 @@ export async function recallByResonance(ctx, query, pre) {
362
360
  // prefixCompletion runs a few lines below and carries the three guards
363
361
  // this tier lacks — unreadable-continuation veto, sub-quantum
364
362
  // continuation, and UNIQUENESS (distinct continuations ⇒ refuse), which
365
- // is exactly what 4,300 competing values must trip. So this is not a
366
- // new rule and not a new threshold: it is deferring a prefix decision to
367
- // the tier that owns it (§2.5, one factored machinery). Byte-strict on
368
- // purpose — a candidate differing by case or punctuation ("what is the
369
- // capital of france" → "What is the capital of France?") is NOT a byte
370
- // prefix, keeps grounding here, and is unaffected.
363
+ // is exactly what 4,300 competing values must trip. So this is not a new
364
+ // rule and not a new threshold: it is deferring a prefix decision to the
365
+ // tier that owns it (match-project.md, one factored machinery).
366
+ // Byte-strict on purpose — a candidate differing by case or punctuation
367
+ // ("what is the capital of france" → "What is the capital of France?") is
368
+ // NOT a byte prefix, keeps grounding here, and is unaffected.
371
369
  const strictPrefix = g !== null &&
372
370
  cBytes.length > query.length &&
373
371
  indexOf(cBytes, query, 0) === 0;
@@ -425,15 +423,15 @@ export async function recallByResonance(ctx, query, pre) {
425
423
  }
426
424
  }
427
425
  }
428
- // The refusal/echo decision. The echo returns a stored form's bytes AS
429
- // the answer — a near-identity claim about the query — and identity-grade
426
+ // The refusal/echo decision. The echo returns a stored form's bytes AS the
427
+ // answer — a near-identity claim about the query — and identity-grade
430
428
  // decisions are never made on an estimated score ("approximate scores may
431
- // rank and propose; they may never decide", §6.2): the RaBitQ estimate
432
- // overshooting the reach bar echoed a WRONG-entity neighbour ("capital of
433
- // Zamunda?" echoed the Armenia fact, observed). The bytes are read
434
- // anyway to be echoed, so the decision uses their EXACT fold: one river
435
- // fold of the top hit, measured in the same query-relative,
436
- // chance-corrected units as the tier above.
429
+ // rank and propose; they may never decide", exact-vs-approximate.md): the
430
+ // RaBitQ estimate overshooting the reach bar echoed a WRONG-entity neighbour
431
+ // ("capital of Zamunda?" echoed the Armenia fact, observed). The bytes are
432
+ // read anyway to be echoed, so the decision uses their EXACT fold: one river
433
+ // fold of the top hit, measured in the same query-relative, chance-corrected
434
+ // units as the tier above.
437
435
  const topBytes = read(ctx, top.id);
438
436
  const exact = topBytes.length > 0
439
437
  ? cosine(queryGist, gistOf(ctx, topBytes))
@@ -2,8 +2,8 @@
2
2
  // bytes (Grounding IV).
3
3
  //
4
4
  // This file is a CONFIGURATION of the shared frame reading in match.ts, not a
5
- // pipeline of its own. The three parts it configures live where §2.5 puts
6
- // them and are reachable by any mechanism:
5
+ // pipeline of its own. The three parts it configures live where
6
+ // match-project.md puts them and are reachable by any mechanism:
7
7
  //
8
8
  // matcher Precomputed.frames() — the frame INVENTORY: which ranked
9
9
  // candidates read as instances of the query's own frame, and
@@ -29,10 +29,10 @@
29
29
  // candidate's continuation UNSUBSTITUTED, so admitting a slot-gap there would
30
30
  // voice the corpus's filler for the asker's referent — the misreference
31
31
  // measured live on the trained store ("How do you say 'flurbish' in French?"
32
- // answered "the way to say hello is \"Bonjour\""). Nor is CAST rewired: its
32
+ // answered "the way to say hello is \"Bonjour\""). Nor is CAST rewired: its
33
33
  // frame gate is WEAVE-local while a slot is COHORT-local, and substituting one
34
- // population for the other is the error §2.7 names. The notion is made
35
- // AVAILABLE, never imposed.
34
+ // population for the other is the error commonality.md names. The notion is
35
+ // made AVAILABLE, never imposed.
36
36
  import { carriesFillers, distinct, follow, substituteAll } from "../match.js";
37
37
  import { dominates } from "../../geometry.js";
38
38
  import { bytesEqual, indexOf } from "../../bytes.js";
@@ -43,16 +43,16 @@ import { rItem, rNode, traceFail } from "../trace.js";
43
43
  * agrees with nothing, so no carriage is attested — the same "two or no
44
44
  * constituent" reading frame-filler's contentRuns applies.
45
45
  *
46
- * THIS IS ALSO THE MECHANISM'S REACH. Evidence comes from the shared top-k
47
- * resonance, so a frame the corpus instantiates only ONCE within k is not
48
- * reachable here. Measured on the trained store: `How do you say 'flurbish'
49
- * in French?` finds one instance of its frame in the top 24 — the rest are
50
- * `How do you make …`, a different frame — so this abstains and recall's
51
- * scaffolding-dominated tier answers with the CORPUS's filler. That
52
- * misreference is recall's, and widening the supply is not the fix: the
53
- * exhaustive √N list recall's refusal path builds costs hundreds of
54
- * milliseconds and this runs before it. Abstaining on thin evidence is the
55
- * honest reading (§2.13). */
46
+ * THIS IS ALSO THE MECHANISM'S REACH. Evidence comes from the shared top-k
47
+ * resonance, so a frame the corpus instantiates only ONCE within k is not
48
+ * reachable here. Measured on the trained store: `How do you say 'flurbish' in
49
+ * French?` finds one instance of its frame in the top 24 — the rest are `How do
50
+ * you make …`, a different frame — so this abstains and recall's
51
+ * scaffolding-dominated tier answers with the CORPUS's filler. That
52
+ * misreference is recall's, and widening the supply is not the fix: the
53
+ * exhaustive √N list recall's refusal path builds costs hundreds of
54
+ * milliseconds and this runs before it. Abstaining on thin evidence is the
55
+ * honest reading (INVARIANTS.md). */
56
56
  const MIN_INSTANCES = 2;
57
57
  /** THE VOICING GATES — this mechanism's own reading of a pairing, applied here
58
58
  * and NOT in the shared matcher.
@@ -123,7 +123,7 @@ function electFrame(inventory, W, queryLen) {
123
123
  let best = [];
124
124
  for (const group of bySignature.values()) {
125
125
  // Ties keep the FIRST group in insertion order, which is resonance rank —
126
- // corpus-determined, like every other tie-break here (§2.1).
126
+ // corpus-determined, like every other tie-break here (determinism.md).
127
127
  if (group.length > best.length)
128
128
  best = group;
129
129
  }
@@ -1,5 +1,6 @@
1
1
  import { Vec } from "../vec.js";
2
2
  import { Sema, Space } from "../sema.js";
3
+ import type { CorpusResult } from "./corpus.js";
3
4
  import { Alphabet } from "../alphabet.js";
4
5
  import { Grid } from "../geometry.js";
5
6
  import { BoundedMap, type Store } from "../store.js";
@@ -49,10 +50,40 @@ import type { AttentionRead, MindContext, Recognition } from "./types.js";
49
50
  export type { AnchorRejectionReason, ClimbConsensusData, ConsensusAnchorTrace, ConsensusReachTrace, ConsensusRegionTrace, CrossRegionTier, JunctionVoteTrace, RegionOutcome, } from "./attention.js";
50
51
  export type { AncestorReach, SaturationReason, SaturationStop, } from "./types.js";
51
52
  import { type CostReport, Meter } from "../meter.js";
53
+ /** A stored pair as TEXT — the text helper's view of {@link CorpusPair}. */
54
+ export interface CorpusTextPair {
55
+ context: string;
56
+ continuation: string;
57
+ contextId: number;
58
+ continuationId: number;
59
+ matchedBytes: number;
60
+ contextTruncated: boolean;
61
+ continuationTruncated: boolean;
62
+ }
63
+ /** {@link CorpusResult} as text, plus the prose for why nothing matched. The
64
+ * byte layer reports a STATE; saying it in words belongs to the text layer. */
65
+ export interface CorpusTextResult {
66
+ query: string;
67
+ pairs: CorpusTextPair[];
68
+ resolved: number;
69
+ reached: number;
70
+ totalContexts: number;
71
+ browsed: boolean;
72
+ note?: string;
73
+ }
52
74
  export interface MindOptions {
53
75
  seed?: number;
54
76
  recallQueryK?: number;
55
77
  haloQueryK?: number;
78
+ /** Items one rationale step may itemise — see {@link MindConfig}. */
79
+ rationaleSampleK?: number;
80
+ /** Corpus-reading capacities and budgets — see {@link MindConfig}. */
81
+ corpusLimitMax?: number;
82
+ corpusClimbs?: number;
83
+ corpusContextsPerClimb?: number;
84
+ corpusSampleProbes?: number;
85
+ corpusPreviewBytes?: number;
86
+ corpusSampleFloorBytes?: number;
56
87
  normalizeEpsilon?: number;
57
88
  cosineEpsilon?: number;
58
89
  geometry?: Partial<import("../config.js").GeometryConfig>;
@@ -64,13 +95,12 @@ export interface MindOptions {
64
95
  /** Factories that receive the {@link ExtensionHost} and return mechanisms. */
65
96
  mechanismFactories?: ((host: import("../extension.js").ExtensionHost) => import("./pipeline-mechanism.js").PipelineMechanism)[];
66
97
  /** Measure the computational usage of every inference call — see
67
- * src/meter.ts. Off by default and free when off (one null check per
68
- * store read); on, each `respond`/`respondTurn` leaves a {@link
69
- * Mind.lastCost} report behind. Counters are deterministic, so two runs
70
- * of the same query on the same store are diffable; the millisecond
71
- * fields are not. Profiling NEVER changes an answer — but note that
72
- * attaching a RATIONALE does (traced responses bypass the ctx memos,
73
- * AGENTS §2.11), so profile without a trace. */
98
+ * src/meter.ts. Off by default and free when off (one null check per store
99
+ * read); on, each `respond`/`respondTurn` leaves a {@link Mind.lastCost}
100
+ * report behind. Counters are deterministic, so two runs of the same query on
101
+ * the same store are diffable; the millisecond fields are not. Profiling
102
+ * NEVER changes an answer — but attaching a RATIONALE does: a traced response
103
+ * bypasses the ctx memos (memoization.md), so profile without a trace. */
74
104
  profile?: boolean;
75
105
  /** Content canonicalizer applied to EVERY response (any modality) for
76
106
  * equivalence-class resolution — see src/canon.ts. Text entry points
@@ -154,6 +184,10 @@ export declare class Mind implements MindContext {
154
184
  * `traverse.ts`'s ONE definition (edge or halo, with its response-scoped
155
185
  * cache). The search holds a bare Store and cannot reach that cache itself,
156
186
  * so it asks through this hook; a bare host keeps its raw-store fallback. */
187
+ /** The canonical identity for the search (see GraphSearchHost). */
188
+ canonResolve(bytes: Uint8Array): number | null;
189
+ /** Feed a search refusal into the rationale (see GraphSearchHost). */
190
+ reportSearch(name: string, parts: ReadonlyArray<Uint8Array>, note: string): void;
157
191
  leadsSomewhere(id: number): boolean;
158
192
  recogniseSpan(bytes: Uint8Array): {
159
193
  sites: ReadonlyArray<Site>;
@@ -168,6 +202,8 @@ export declare class Mind implements MindContext {
168
202
  * with the most distributional evidence (highest `prevOf` count — the
169
203
  * structural manifestation of its halo). When evidence is equal the
170
204
  * first-inserted edge wins. */
205
+ /** See {@link GraphSearchHost.contentCuts}. */
206
+ contentCuts(bytes: Uint8Array): readonly number[];
171
207
  chooseNext(node: number): number | undefined;
172
208
  constructor(opts?: MindOptions);
173
209
  constructor(cfg: MindConfig, store: Store, _fromStore: true);
@@ -250,6 +286,24 @@ export declare class Mind implements MindContext {
250
286
  * as one form, provided the store's canon index is built
251
287
  * ({@link buildCanonIndex}). */
252
288
  respondText(input: string, inspectRationale?: InspectRationale): Promise<string>;
289
+ /** Which stored notes does this query REACH? BYTES in, BYTES out — this
290
+ * method has no notion of text or encoding; the text case is
291
+ * {@link searchCorpusText}, which is one caller of this.
292
+ *
293
+ * Exact content addressing through the machinery an answer already uses
294
+ * (see src/mind/corpus.ts): the query's recognised sites are the resolved
295
+ * subtrees, the climb goes up from the biggest, and a result is a context
296
+ * that carries a learnt continuation. Nothing is written and nothing is
297
+ * indexed. */
298
+ searchCorpus(queryBytes: Uint8Array, limit?: number): CorpusResult;
299
+ /** Browse real pairs. Deterministic: `from` is the caller's own offset in
300
+ * [0,1), so browsing twice with different offsets shows different notes
301
+ * without a random draw. */
302
+ sampleCorpus(limit?: number, from?: number): CorpusResult;
303
+ /** The TEXT case of {@link searchCorpus}: encode, search, decode. The search
304
+ * itself exists once, in the byte layer above; only the rendering lives
305
+ * here, with the rest of this class's text modality. */
306
+ searchCorpusText(query: string, limit?: number): CorpusTextResult;
253
307
  /** Begin a new conversation, optionally restoring from a previously-saved
254
308
  * {@link ConversationState}. The returned handle is required for
255
309
  * {@link respondTurn} and {@link endConversation}.
@@ -9,8 +9,9 @@
9
9
  // Architecture: 4 primitives × 2 patterns = all inference.
10
10
  // Implementation split across src/mind/*.ts — this file assembles the Mind class.
11
11
  import { makeKeyring, rng, setVecConfig } from "../vec.js";
12
+ import { sampleCorpus, searchCorpus } from "./corpus.js";
12
13
  import { Alphabet } from "../alphabet.js";
13
- import { contentFoldIncremental, reachThreshold, } from "../geometry.js";
14
+ import { contentBoundaries, contentFoldIncremental, reachThreshold, } from "../geometry.js";
14
15
  import { BoundedMap } from "../store.js";
15
16
  import { SQliteStore } from "../store-sqlite.js";
16
17
  import { resolveConfig } from "../config.js";
@@ -19,7 +20,7 @@ import { bytesEqual, concat2 } from "../bytes.js";
19
20
  import { GraphSearch, } from "./graph-search.js";
20
21
  import { Alu } from "../alu/src/index.js";
21
22
  import { decodeText, Rationale, } from "./rationale.js";
22
- import { gistOf, inputBytes, perceive as perceiveImpl, perceiveKey, resolve as resolveImpl, } from "./primitives.js";
23
+ import { canonResolve as canonResolveImpl, gistOf, inputBytes, perceive as perceiveImpl, perceiveKey, resolve as resolveImpl, } from "./primitives.js";
23
24
  import { chooseNext, edgeAncestors as edgeAncestorsFn, invalidateStructuralCaches, leadsSomewhere, } from "./traverse.js";
24
25
  import { invalidateJunctionCache } from "./junction.js";
25
26
  import { follow } from "./match.js";
@@ -33,6 +34,21 @@ import { rItem } from "./trace.js";
33
34
  // The work meter is exported from src/index.ts (via src/meter.ts) — the one
34
35
  // definition; the Mind only consumes it.
35
36
  import { Meter } from "../meter.js";
37
+ /** What the text helper says when the byte layer reports a miss. */
38
+ const CORPUS_NOTE = {
39
+ "nothing-resolved": "No trained note sits above the parts of that text the mind recognised. " +
40
+ "It addresses content exactly, so try wording closer to something it was " +
41
+ "actually given — or browse the examples instead.",
42
+ "no-continuations": "That text reaches stored nodes, but none of them carries a learnt " +
43
+ "continuation.",
44
+ };
45
+ /** UTF-8 of bytes for display: reuse {@link decodeText} (the mind's own text
46
+ * conversion), then drop the replacement character a byte-boundary cut leaves
47
+ * behind. Much of a real corpus is non-Latin, so that trailing U+FFFD is the
48
+ * common case, not an exotic one — and it is the ONLY thing added here. */
49
+ function previewCorpusText(bytes) {
50
+ return decodeText(bytes).replace(/\uFFFD+$/, "").replace(/\s+/g, " ").trim();
51
+ }
36
52
  // ═══════════════════════════════════════════════════════════════════════════
37
53
  // THE MIND
38
54
  // ═══════════════════════════════════════════════════════════════════════════
@@ -116,6 +132,14 @@ export class Mind {
116
132
  * `traverse.ts`'s ONE definition (edge or halo, with its response-scoped
117
133
  * cache). The search holds a bare Store and cannot reach that cache itself,
118
134
  * so it asks through this hook; a bare host keeps its raw-store fallback. */
135
+ /** The canonical identity for the search (see GraphSearchHost). */
136
+ canonResolve(bytes) {
137
+ return canonResolveImpl(this, bytes);
138
+ }
139
+ /** Feed a search refusal into the rationale (see GraphSearchHost). */
140
+ reportSearch(name, parts, note) {
141
+ this.trace?.step(name, parts.map((b) => rItem(b)), [], note);
142
+ }
119
143
  leadsSomewhere(id) {
120
144
  return leadsSomewhere(this, id);
121
145
  }
@@ -136,6 +160,10 @@ export class Mind {
136
160
  * with the most distributional evidence (highest `prevOf` count — the
137
161
  * structural manifestation of its halo). When evidence is equal the
138
162
  * first-inserted edge wins. */
163
+ /** See {@link GraphSearchHost.contentCuts}. */
164
+ contentCuts(bytes) {
165
+ return contentBoundaries(this.space, bytes);
166
+ }
139
167
  chooseNext(node) {
140
168
  return chooseNext(this, node, this._edgeGuide);
141
169
  }
@@ -411,6 +439,48 @@ export class Mind {
411
439
  const r = await this.respond(input, inspectRationale);
412
440
  return decodeText(r.bytes);
413
441
  }
442
+ // ── Reading the trained memory back ─────────────────────────────────────
443
+ /** Which stored notes does this query REACH? BYTES in, BYTES out — this
444
+ * method has no notion of text or encoding; the text case is
445
+ * {@link searchCorpusText}, which is one caller of this.
446
+ *
447
+ * Exact content addressing through the machinery an answer already uses
448
+ * (see src/mind/corpus.ts): the query's recognised sites are the resolved
449
+ * subtrees, the climb goes up from the biggest, and a result is a context
450
+ * that carries a learnt continuation. Nothing is written and nothing is
451
+ * indexed. */
452
+ searchCorpus(queryBytes, limit) {
453
+ return searchCorpus(this, queryBytes, limit);
454
+ }
455
+ /** Browse real pairs. Deterministic: `from` is the caller's own offset in
456
+ * [0,1), so browsing twice with different offsets shows different notes
457
+ * without a random draw. */
458
+ sampleCorpus(limit, from) {
459
+ return sampleCorpus(this, limit, from);
460
+ }
461
+ /** The TEXT case of {@link searchCorpus}: encode, search, decode. The search
462
+ * itself exists once, in the byte layer above; only the rendering lives
463
+ * here, with the rest of this class's text modality. */
464
+ searchCorpusText(query, limit) {
465
+ const result = this.searchCorpus(new TextEncoder().encode(query), limit);
466
+ return {
467
+ query,
468
+ pairs: result.pairs.map((p) => ({
469
+ context: previewCorpusText(p.context),
470
+ continuation: previewCorpusText(p.continuation),
471
+ contextId: p.contextId,
472
+ continuationId: p.continuationId,
473
+ matchedBytes: p.matchedBytes,
474
+ contextTruncated: p.contextTruncated,
475
+ continuationTruncated: p.continuationTruncated,
476
+ })),
477
+ resolved: result.resolved,
478
+ reached: result.reached,
479
+ totalContexts: result.totalContexts,
480
+ browsed: result.browsed,
481
+ note: result.miss === "matched" ? undefined : CORPUS_NOTE[result.miss],
482
+ };
483
+ }
414
484
  // ── Conversation API ────────────────────────────────────────────────────
415
485
  /** Begin a new conversation, optionally restoring from a previously-saved
416
486
  * {@link ConversationState}. The returned handle is required for
@@ -69,14 +69,16 @@ export declare class Precomputed {
69
69
  * ({@link FrameInstance}). The one place the engine represents "a position
70
70
  * whose occupant comes from the context rather than the corpus".
71
71
  *
72
- * AN INVENTORY, NOT AN ELECTION. It reports every pairing and elects no
73
- * frame, deliberately: a slot is a property of a PAIRING, not of the query,
74
- * and different candidates put slots in different places. Committing to one
75
- * reading here would push whichever consumer asked first onto everyone else
76
- * — the market's decoupling (§2.6) broken from inside the shared container,
77
- * and the population error §2.7 names. Each consumer groups and commits
78
- * for its own question; reference elects the modal slot signature, and a
79
- * consumer wanting a different reading is not fighting this one.
72
+ * AN INVENTORY, NOT AN ELECTION. It reports every pairing and elects no
73
+ * frame,
74
+ * deliberately: a slot is a property of a PAIRING, not of the query, and
75
+ * different candidates put slots in different places. Committing to one
76
+ * reading here would push whichever consumer asked first onto everyone else —
77
+ * the market's decoupling (mechanism-market.md) broken from inside the shared
78
+ * container, and the population error commonality.md names. Each consumer
79
+ * groups and commits for its own question; reference elects the modal slot
80
+ * signature, and a consumer wanting a different reading is not fighting this
81
+ * one.
80
82
  *
81
83
  * NO LICENCE EITHER. Knowing a span is variable is safe for every consumer
82
84
  * — it can only improve an alignment. Knowing one may be VOICED through is
@@ -5,7 +5,7 @@
5
5
  // a list of PipelineMechanism objects — it never imports a mechanism-specific
6
6
  // type and never has a special-case branch for any mechanism.
7
7
  //
8
- // The four constraints of the free-will architecture (§14.5):
8
+ // The four constraints of the free-will architecture (mechanism-market.md):
9
9
  // 1. DECOUPLING — mechanisms import nothing from each other or from pipeline.
10
10
  // 2. DECLARED COMPETENCE — floor() returns null when impossible, a number when
11
11
  // possible. Binary, auditable, no learned scores.
@@ -128,31 +128,33 @@ export class Precomputed {
128
128
  resonance() {
129
129
  return this._resonance ??= this.shared("resonance", () => this.ctx.store.resonate(this.guide, this.k));
130
130
  }
131
- // REMOVED — the WIDE exhaustive-√N resonance list (`wideResonance`). It ran
131
+ // REMOVED — the WIDE exhaustive-√N resonance list (`wideResonance`). It ran
132
132
  // `resonate(guide, √N, exhaustive=true)` whenever the top hit cleared
133
- // conceptThreshold, so consumers could look "past the top-k". Every consumer
133
+ // conceptThreshold, so consumers could look "past the top-k". Every consumer
134
134
  // only ever needed ≤ 2·recallQueryK proposals (the substitution bridge's own
135
135
  // candidate cap) or a content-addressed answer (prefix completion's
136
- // formsOpenedBy), and every proposal is byte-verified downstream (§2.3), so
137
- // the exhaustive scan bought recall at O(index) cost for an O(k) need —
138
- // measured: 244K annVectorReads per refusing query, ~1.5 s, every answer
139
- // byte-identical to a top-k read. The two consumers now read `resonance()`
140
- // (the one top-k read) and the write side's window index respectively — see
141
- // recall.ts and prefix-completion.ts.
136
+ // formsOpenedBy), and every proposal is byte-verified downstream
137
+ // (exact-vs-approximate.md), so the exhaustive scan bought recall at O(index)
138
+ // cost for an O(k) need — measured: 244K annVectorReads per refusing query,
139
+ // ~1.5 s, every answer byte-identical to a top-k read. The two consumers now
140
+ // read `resonance()` (the one top-k read) and the write side's window index
141
+ // respectively — see recall.ts and prefix-completion.ts.
142
142
  _frames;
143
143
  /** THE FRAME INVENTORY — every ranked candidate that reads as an instance of
144
144
  * the same frame as the query, each with the query spans it leaves VARIABLE
145
145
  * ({@link FrameInstance}). The one place the engine represents "a position
146
146
  * whose occupant comes from the context rather than the corpus".
147
147
  *
148
- * AN INVENTORY, NOT AN ELECTION. It reports every pairing and elects no
149
- * frame, deliberately: a slot is a property of a PAIRING, not of the query,
150
- * and different candidates put slots in different places. Committing to one
151
- * reading here would push whichever consumer asked first onto everyone else
152
- * — the market's decoupling (§2.6) broken from inside the shared container,
153
- * and the population error §2.7 names. Each consumer groups and commits
154
- * for its own question; reference elects the modal slot signature, and a
155
- * consumer wanting a different reading is not fighting this one.
148
+ * AN INVENTORY, NOT AN ELECTION. It reports every pairing and elects no
149
+ * frame,
150
+ * deliberately: a slot is a property of a PAIRING, not of the query, and
151
+ * different candidates put slots in different places. Committing to one
152
+ * reading here would push whichever consumer asked first onto everyone else —
153
+ * the market's decoupling (mechanism-market.md) broken from inside the shared
154
+ * container, and the population error commonality.md names. Each consumer
155
+ * groups and commits for its own question; reference elects the modal slot
156
+ * signature, and a consumer wanting a different reading is not fighting this
157
+ * one.
156
158
  *
157
159
  * NO LICENCE EITHER. Knowing a span is variable is safe for every consumer
158
160
  * — it can only improve an alignment. Knowing one may be VOICED through is
@@ -168,9 +170,10 @@ export class Precomputed {
168
170
  const capBytes = this.query.length * W;
169
171
  const out = [];
170
172
  for (const h of await this.resonance()) {
171
- // REJECT BY LENGTH BEFORE RECONSTRUCTING (§2.8): `contentLen` is an
172
- // indexed read, `bytesPrefix` rebuilds a subtree. ONLY the phrase-scale
173
- // cap is applied — it is a bounded-read discipline, not a judgement.
173
+ // REJECT BY LENGTH BEFORE RECONSTRUCTING (bounded-reads.md):
174
+ // `contentLen` is an indexed read, `bytesPrefix` rebuilds a subtree.
175
+ // ONLY the phrase-scale cap is applied — it is a bounded-read
176
+ // discipline, not a judgement.
174
177
  //
175
178
  // A LOWER bound was here too (`dominates(len, query.length)`, on the
176
179
  // reasoning that a candidate shorter than half the query cannot supply
@@ -458,7 +461,8 @@ function computeWeave(ctx, query, pre, climb) {
458
461
  // IDF — gates the aligner has no equivalent of.
459
462
  //
460
463
  // So the climb PROPOSES the pairing (which structure, which query span) and
461
- // bytes DECIDE its terms (§2.3). Three gates, each one measured:
464
+ // bytes DECIDE its terms (exact-vs-approximate.md). Three gates, each one
465
+ // measured:
462
466
  //
463
467
  // • it may only take query bytes NO literal run claimed. Run inline with
464
468
  // phase 1 this did the opposite of "exact decides" — a higher-ranked
@@ -38,15 +38,15 @@ export interface NarrowDecisionData {
38
38
  margin: number;
39
39
  }
40
40
  /** Structured payload of the "regimePrediction" rationale step — the R8
41
- * observation exposed as data. After the first mechanism (cover, which §2.6
42
- * runs first) grounds or abstains, the market's whole outcome is already
43
- * determined by the one cost ladder: the consensus climb runs exactly when
44
- * `worthRunning(2 * STEP)` is true — CAST (floor 2·STEP) is the cheapest
45
- * mechanism that first-touches it, and confluence (3·STEP) / extraction
46
- * (CONCEPT+STEP) are only reached after CAST is. An incumbent at or below
47
- * that floor prunes CAST and, with it, the climb (retrieval); anything above
48
- * — or no incumbent — runs the full market and the climb (composition).
49
- * Purely observational; never read by inference. */
41
+ * observation exposed as data. After the first mechanism (cover, which
42
+ * mechanism-market.md runs first) grounds or abstains, the market's whole
43
+ * outcome is already determined by the one cost ladder: the consensus climb
44
+ * runs exactly when `worthRunning(2 * STEP)` is true — CAST (floor 2·STEP) is
45
+ * the cheapest mechanism that first-touches it, and confluence (3·STEP) /
46
+ * extraction (CONCEPT+STEP) are only reached after CAST is. An incumbent at or
47
+ * below that floor prunes CAST and, with it, the climb (retrieval); anything
48
+ * above — or no incumbent — runs the full market and the climb (composition).
49
+ * Purely observational; never read by inference. */
50
50
  export interface RegimePredictionData {
51
51
  version: 1;
52
52
  /** retrieval | composition — the two regimes R1 measured as a ~100× cost