@hviana/sema 0.8.0 → 0.8.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (119) hide show
  1. package/AGENTS.md +22 -1
  2. package/DATASETS.md +1 -1
  3. package/dist/example/train_base/config.js +2 -2
  4. package/dist/example/train_base/corpora/massive.js +1 -1
  5. package/dist/example/train_base/readers.js +1 -1
  6. package/dist/src/config.d.ts +17 -0
  7. package/dist/src/config.js +18 -0
  8. package/dist/src/geometry.d.ts +10 -10
  9. package/dist/src/geometry.js +25 -24
  10. package/dist/src/meter.d.ts +29 -12
  11. package/dist/src/meter.js +58 -14
  12. package/dist/src/mind/attention.js +12 -12
  13. package/dist/src/mind/bridge.d.ts +8 -8
  14. package/dist/src/mind/bridge.js +33 -32
  15. package/dist/src/mind/corpus.d.ts +40 -0
  16. package/dist/src/mind/corpus.js +149 -0
  17. package/dist/src/mind/graph-search.d.ts +7 -8
  18. package/dist/src/mind/graph-search.js +244 -32
  19. package/dist/src/mind/index.d.ts +3 -1
  20. package/dist/src/mind/index.js +1 -0
  21. package/dist/src/mind/junction.d.ts +1 -1
  22. package/dist/src/mind/junction.js +8 -8
  23. package/dist/src/mind/learning.js +36 -35
  24. package/dist/src/mind/match.d.ts +8 -3
  25. package/dist/src/mind/match.js +156 -71
  26. package/dist/src/mind/mechanisms/cast.js +18 -2
  27. package/dist/src/mind/mechanisms/cover.js +19 -12
  28. package/dist/src/mind/mechanisms/prefix-completion.js +24 -24
  29. package/dist/src/mind/mechanisms/recall.js +38 -40
  30. package/dist/src/mind/mechanisms/reference.js +16 -16
  31. package/dist/src/mind/mind.d.ts +61 -7
  32. package/dist/src/mind/mind.js +72 -2
  33. package/dist/src/mind/pipeline-mechanism.d.ts +10 -8
  34. package/dist/src/mind/pipeline-mechanism.js +25 -21
  35. package/dist/src/mind/pipeline.d.ts +9 -9
  36. package/dist/src/mind/pipeline.js +49 -29
  37. package/dist/src/mind/primitives.d.ts +5 -5
  38. package/dist/src/mind/primitives.js +5 -5
  39. package/dist/src/mind/reasoning.d.ts +5 -1
  40. package/dist/src/mind/reasoning.js +54 -1
  41. package/dist/src/mind/recognition.d.ts +14 -13
  42. package/dist/src/mind/recognition.js +23 -23
  43. package/dist/src/mind/resonance.js +21 -21
  44. package/dist/src/mind/traverse.d.ts +54 -52
  45. package/dist/src/mind/traverse.js +83 -73
  46. package/dist/src/mind/types.d.ts +26 -4
  47. package/dist/src/store.d.ts +12 -12
  48. package/dist/src/store.js +12 -12
  49. package/docs/INDEX.md +2 -2
  50. package/docs/architecture/exact-vs-approximate.md +2 -1
  51. package/docs/architecture/fold-contract.md +1 -1
  52. package/docs/failures/tempting-but-wrong.md +33 -5
  53. package/docs/harness/gates.md +7 -7
  54. package/example/train_base/config.ts +2 -2
  55. package/example/train_base/corpora/massive.ts +1 -1
  56. package/example/train_base/readers.ts +1 -1
  57. package/jsr.json +1 -1
  58. package/package.json +1 -1
  59. package/src/config.ts +35 -0
  60. package/src/geometry.ts +25 -24
  61. package/src/meter.ts +61 -14
  62. package/src/mind/attention.ts +12 -12
  63. package/src/mind/bridge.ts +33 -32
  64. package/src/mind/corpus.ts +202 -0
  65. package/src/mind/graph-search.ts +261 -31
  66. package/src/mind/index.ts +8 -1
  67. package/src/mind/junction.ts +8 -8
  68. package/src/mind/learning.ts +36 -35
  69. package/src/mind/match.ts +163 -73
  70. package/src/mind/mechanisms/cast.ts +17 -1
  71. package/src/mind/mechanisms/cover.ts +18 -12
  72. package/src/mind/mechanisms/prefix-completion.ts +24 -24
  73. package/src/mind/mechanisms/recall.ts +38 -40
  74. package/src/mind/mechanisms/reference.ts +16 -16
  75. package/src/mind/mind.ts +129 -7
  76. package/src/mind/pipeline-mechanism.ts +25 -21
  77. package/src/mind/pipeline.ts +63 -38
  78. package/src/mind/primitives.ts +5 -5
  79. package/src/mind/reasoning.ts +55 -0
  80. package/src/mind/recognition.ts +23 -23
  81. package/src/mind/resonance.ts +21 -21
  82. package/src/mind/traverse.ts +83 -73
  83. package/src/mind/types.ts +30 -4
  84. package/src/store.ts +20 -20
  85. package/test/08-storage.test.mjs +1 -1
  86. package/test/100-complete-grounding-trace.test.mjs +109 -0
  87. package/test/101-alignment-gap-bound.test.mjs +106 -0
  88. package/test/102-production-composes-at-scale.test.mjs +110 -0
  89. package/test/103-alignment-gap-budget.test.mjs +89 -0
  90. package/test/104-composition-is-reported.test.mjs +90 -0
  91. package/test/105-derive-through-reports-its-refusal.test.mjs +113 -0
  92. package/test/106-the-join-fires.test.mjs +94 -0
  93. package/test/107-the-join-is-counted.test.mjs +81 -0
  94. package/test/108-the-join-chains.test.mjs +78 -0
  95. package/test/109-the-pivot-is-counted.test.mjs +60 -0
  96. package/test/110-the-reasoner-stops-when-the-question-is-answered.test.mjs +91 -0
  97. package/test/111-the-cover-assembly-is-counted.test.mjs +74 -0
  98. package/test/112-the-exploration-does-not-grow-with-the-hub.test.mjs +89 -0
  99. package/test/113-the-rationale-payload-is-bounded.test.mjs +84 -0
  100. package/test/114-alignment-budget-is-per-sweep.test.mjs +93 -0
  101. package/test/116-the-extension-is-gated-by-the-pipelines-own-remainder.test.mjs +100 -0
  102. package/test/117-corpus-search.test.mjs +171 -0
  103. package/test/14-scaling.test.mjs +10 -7
  104. package/test/35-prefix-edge.test.mjs +1 -1
  105. package/test/40-choosenext-scale-guard.test.mjs +16 -17
  106. package/test/56-bridge-identity-admission.test.mjs +6 -6
  107. package/test/70-prefix-completion.test.mjs +4 -3
  108. package/test/72-prefix-candidate-supply.test.mjs +3 -3
  109. package/test/73-scaffolding-only-bridge-abstains.test.mjs +6 -6
  110. package/test/75-multiturn-context-optimisation.test.mjs +5 -5
  111. package/test/76-reference-binding.test.mjs +6 -1
  112. package/test/84-composed-answer-honesty.test.mjs +5 -6
  113. package/test/88-dependency-footprint.test.mjs +1 -1
  114. package/test/89-completion-recursion.test.mjs +47 -19
  115. package/test/90-connector-read-cap.test.mjs +10 -8
  116. package/test/93-regime-prediction.test.mjs +10 -10
  117. package/test/94-cross-region-budget.test.mjs +2 -2
  118. package/test/95-wide-resonance-removed.test.mjs +8 -7
  119. package/test/96-bytes-walk-termination.test.mjs +3 -3
@@ -2,8 +2,8 @@
2
2
  // bytes (Grounding IV).
3
3
  //
4
4
  // This file is a CONFIGURATION of the shared frame reading in match.ts, not a
5
- // pipeline of its own. The three parts it configures live where §2.5 puts
6
- // them and are reachable by any mechanism:
5
+ // pipeline of its own. The three parts it configures live where
6
+ // match-project.md puts them and are reachable by any mechanism:
7
7
  //
8
8
  // matcher Precomputed.frames() — the frame INVENTORY: which ranked
9
9
  // candidates read as instances of the query's own frame, and
@@ -29,10 +29,10 @@
29
29
  // candidate's continuation UNSUBSTITUTED, so admitting a slot-gap there would
30
30
  // voice the corpus's filler for the asker's referent — the misreference
31
31
  // measured live on the trained store ("How do you say 'flurbish' in French?"
32
- // answered "the way to say hello is \"Bonjour\""). Nor is CAST rewired: its
32
+ // answered "the way to say hello is \"Bonjour\""). Nor is CAST rewired: its
33
33
  // frame gate is WEAVE-local while a slot is COHORT-local, and substituting one
34
- // population for the other is the error §2.7 names. The notion is made
35
- // AVAILABLE, never imposed.
34
+ // population for the other is the error commonality.md names. The notion is
35
+ // made AVAILABLE, never imposed.
36
36
 
37
37
  import type { MindContext } from "../types.js";
38
38
  import type { FrameInstance } from "../match.js";
@@ -52,16 +52,16 @@ import { rItem, rNode, traceFail } from "../trace.js";
52
52
  * agrees with nothing, so no carriage is attested — the same "two or no
53
53
  * constituent" reading frame-filler's contentRuns applies.
54
54
  *
55
- * THIS IS ALSO THE MECHANISM'S REACH. Evidence comes from the shared top-k
56
- * resonance, so a frame the corpus instantiates only ONCE within k is not
57
- * reachable here. Measured on the trained store: `How do you say 'flurbish'
58
- * in French?` finds one instance of its frame in the top 24 — the rest are
59
- * `How do you make …`, a different frame — so this abstains and recall's
60
- * scaffolding-dominated tier answers with the CORPUS's filler. That
61
- * misreference is recall's, and widening the supply is not the fix: the
62
- * exhaustive √N list recall's refusal path builds costs hundreds of
63
- * milliseconds and this runs before it. Abstaining on thin evidence is the
64
- * honest reading (§2.13). */
55
+ * THIS IS ALSO THE MECHANISM'S REACH. Evidence comes from the shared top-k
56
+ * resonance, so a frame the corpus instantiates only ONCE within k is not
57
+ * reachable here. Measured on the trained store: `How do you say 'flurbish' in
58
+ * French?` finds one instance of its frame in the top 24 — the rest are `How do
59
+ * you make …`, a different frame — so this abstains and recall's
60
+ * scaffolding-dominated tier answers with the CORPUS's filler. That
61
+ * misreference is recall's, and widening the supply is not the fix: the
62
+ * exhaustive √N list recall's refusal path builds costs hundreds of
63
+ * milliseconds and this runs before it. Abstaining on thin evidence is the
64
+ * honest reading (INVARIANTS.md). */
65
65
  const MIN_INSTANCES = 2;
66
66
 
67
67
  /** THE VOICING GATES — this mechanism's own reading of a pairing, applied here
@@ -135,7 +135,7 @@ function electFrame(
135
135
  let best: FrameInstance[] = [];
136
136
  for (const group of bySignature.values()) {
137
137
  // Ties keep the FIRST group in insertion order, which is resonance rank —
138
- // corpus-determined, like every other tie-break here (§2.1).
138
+ // corpus-determined, like every other tie-break here (determinism.md).
139
139
  if (group.length > best.length) best = group;
140
140
  }
141
141
  return best;
package/src/mind/mind.ts CHANGED
@@ -11,9 +11,12 @@
11
11
 
12
12
  import { cosine, makeKeyring, rng, setVecConfig, Vec } from "../vec.js";
13
13
  import { bindSeat, fold, Sema, Space } from "../sema.js";
14
+ import { sampleCorpus, searchCorpus } from "./corpus.js";
15
+ import type { CorpusPair, CorpusResult } from "./corpus.js";
14
16
  import { Alphabet } from "../alphabet.js";
15
17
  import {
16
18
  bytesToTree,
19
+ contentBoundaries,
17
20
  contentFoldIncremental,
18
21
  Grid,
19
22
  gridToTree,
@@ -146,6 +149,7 @@ interface ConversationData {
146
149
  import type { AttentionRead, MindContext, Recognition } from "./types.js";
147
150
  import { changedNodes, liftAnswer, spliceAll } from "./types.js";
148
151
  import {
152
+ canonResolve as canonResolveImpl,
149
153
  foldTree,
150
154
  gistOf,
151
155
  inputBytes,
@@ -194,10 +198,61 @@ import { type CostReport, Meter } from "../meter.js";
194
198
 
195
199
  // ── MindOptions ───────────────────────────────────────────────────────────
196
200
 
201
+ /** A stored pair as TEXT — the text helper's view of {@link CorpusPair}. */
202
+ export interface CorpusTextPair {
203
+ context: string;
204
+ continuation: string;
205
+ contextId: number;
206
+ continuationId: number;
207
+ matchedBytes: number;
208
+ contextTruncated: boolean;
209
+ continuationTruncated: boolean;
210
+ }
211
+
212
+ /** {@link CorpusResult} as text, plus the prose for why nothing matched. The
213
+ * byte layer reports a STATE; saying it in words belongs to the text layer. */
214
+ export interface CorpusTextResult {
215
+ query: string;
216
+ pairs: CorpusTextPair[];
217
+ resolved: number;
218
+ reached: number;
219
+ totalContexts: number;
220
+ browsed: boolean;
221
+ note?: string;
222
+ }
223
+
224
+ /** What the text helper says when the byte layer reports a miss. */
225
+ const CORPUS_NOTE: Record<string, string> = {
226
+ "nothing-resolved":
227
+ "No trained note sits above the parts of that text the mind recognised. " +
228
+ "It addresses content exactly, so try wording closer to something it was " +
229
+ "actually given — or browse the examples instead.",
230
+ "no-continuations":
231
+ "That text reaches stored nodes, but none of them carries a learnt " +
232
+ "continuation.",
233
+ };
234
+
235
+ /** UTF-8 of bytes for display: reuse {@link decodeText} (the mind's own text
236
+ * conversion), then drop the replacement character a byte-boundary cut leaves
237
+ * behind. Much of a real corpus is non-Latin, so that trailing U+FFFD is the
238
+ * common case, not an exotic one — and it is the ONLY thing added here. */
239
+ function previewCorpusText(bytes: Uint8Array): string {
240
+ return decodeText(bytes).replace(/\uFFFD+$/, "").replace(/\s+/g, " ").trim();
241
+ }
242
+
197
243
  export interface MindOptions {
198
244
  seed?: number;
199
245
  recallQueryK?: number;
200
246
  haloQueryK?: number;
247
+ /** Items one rationale step may itemise — see {@link MindConfig}. */
248
+ rationaleSampleK?: number;
249
+ /** Corpus-reading capacities and budgets — see {@link MindConfig}. */
250
+ corpusLimitMax?: number;
251
+ corpusClimbs?: number;
252
+ corpusContextsPerClimb?: number;
253
+ corpusSampleProbes?: number;
254
+ corpusPreviewBytes?: number;
255
+ corpusSampleFloorBytes?: number;
201
256
  normalizeEpsilon?: number;
202
257
  cosineEpsilon?: number;
203
258
  geometry?: Partial<import("../config.js").GeometryConfig>;
@@ -211,13 +266,12 @@ export interface MindOptions {
211
266
  host: import("../extension.js").ExtensionHost,
212
267
  ) => import("./pipeline-mechanism.js").PipelineMechanism)[];
213
268
  /** Measure the computational usage of every inference call — see
214
- * src/meter.ts. Off by default and free when off (one null check per
215
- * store read); on, each `respond`/`respondTurn` leaves a {@link
216
- * Mind.lastCost} report behind. Counters are deterministic, so two runs
217
- * of the same query on the same store are diffable; the millisecond
218
- * fields are not. Profiling NEVER changes an answer — but note that
219
- * attaching a RATIONALE does (traced responses bypass the ctx memos,
220
- * AGENTS §2.11), so profile without a trace. */
269
+ * src/meter.ts. Off by default and free when off (one null check per store
270
+ * read); on, each `respond`/`respondTurn` leaves a {@link Mind.lastCost}
271
+ * report behind. Counters are deterministic, so two runs of the same query on
272
+ * the same store are diffable; the millisecond fields are not. Profiling
273
+ * NEVER changes an answer — but attaching a RATIONALE does: a traced response
274
+ * bypasses the ctx memos (memoization.md), so profile without a trace. */
221
275
  profile?: boolean;
222
276
  /** Content canonicalizer applied to EVERY response (any modality) for
223
277
  * equivalence-class resolution — see src/canon.ts. Text entry points
@@ -348,6 +402,20 @@ export class Mind implements MindContext {
348
402
  * `traverse.ts`'s ONE definition (edge or halo, with its response-scoped
349
403
  * cache). The search holds a bare Store and cannot reach that cache itself,
350
404
  * so it asks through this hook; a bare host keeps its raw-store fallback. */
405
+ /** The canonical identity for the search (see GraphSearchHost). */
406
+ canonResolve(bytes: Uint8Array): number | null {
407
+ return canonResolveImpl(this, bytes);
408
+ }
409
+
410
+ /** Feed a search refusal into the rationale (see GraphSearchHost). */
411
+ reportSearch(
412
+ name: string,
413
+ parts: ReadonlyArray<Uint8Array>,
414
+ note: string,
415
+ ): void {
416
+ this.trace?.step(name, parts.map((b) => rItem(b)), [], note);
417
+ }
418
+
351
419
  leadsSomewhere(id: number): boolean {
352
420
  return leadsSomewhere(this, id);
353
421
  }
@@ -375,6 +443,11 @@ export class Mind implements MindContext {
375
443
  * with the most distributional evidence (highest `prevOf` count — the
376
444
  * structural manifestation of its halo). When evidence is equal the
377
445
  * first-inserted edge wins. */
446
+ /** See {@link GraphSearchHost.contentCuts}. */
447
+ contentCuts(bytes: Uint8Array): readonly number[] {
448
+ return contentBoundaries(this.space, bytes);
449
+ }
450
+
378
451
  chooseNext(node: number): number | undefined {
379
452
  return chooseNext(this, node, this._edgeGuide);
380
453
  }
@@ -738,6 +811,55 @@ export class Mind implements MindContext {
738
811
  return decodeText(r.bytes);
739
812
  }
740
813
 
814
+ // ── Reading the trained memory back ─────────────────────────────────────
815
+
816
+ /** Which stored notes does this query REACH? BYTES in, BYTES out — this
817
+ * method has no notion of text or encoding; the text case is
818
+ * {@link searchCorpusText}, which is one caller of this.
819
+ *
820
+ * Exact content addressing through the machinery an answer already uses
821
+ * (see src/mind/corpus.ts): the query's recognised sites are the resolved
822
+ * subtrees, the climb goes up from the biggest, and a result is a context
823
+ * that carries a learnt continuation. Nothing is written and nothing is
824
+ * indexed. */
825
+ searchCorpus(queryBytes: Uint8Array, limit?: number): CorpusResult {
826
+ return searchCorpus(this, queryBytes, limit);
827
+ }
828
+
829
+ /** Browse real pairs. Deterministic: `from` is the caller's own offset in
830
+ * [0,1), so browsing twice with different offsets shows different notes
831
+ * without a random draw. */
832
+ sampleCorpus(limit?: number, from?: number): CorpusResult {
833
+ return sampleCorpus(this, limit, from);
834
+ }
835
+
836
+ /** The TEXT case of {@link searchCorpus}: encode, search, decode. The search
837
+ * itself exists once, in the byte layer above; only the rendering lives
838
+ * here, with the rest of this class's text modality. */
839
+ searchCorpusText(query: string, limit?: number): CorpusTextResult {
840
+ const result = this.searchCorpus(
841
+ new TextEncoder().encode(query),
842
+ limit,
843
+ );
844
+ return {
845
+ query,
846
+ pairs: result.pairs.map((p: CorpusPair): CorpusTextPair => ({
847
+ context: previewCorpusText(p.context),
848
+ continuation: previewCorpusText(p.continuation),
849
+ contextId: p.contextId,
850
+ continuationId: p.continuationId,
851
+ matchedBytes: p.matchedBytes,
852
+ contextTruncated: p.contextTruncated,
853
+ continuationTruncated: p.continuationTruncated,
854
+ })),
855
+ resolved: result.resolved,
856
+ reached: result.reached,
857
+ totalContexts: result.totalContexts,
858
+ browsed: result.browsed,
859
+ note: result.miss === "matched" ? undefined : CORPUS_NOTE[result.miss],
860
+ };
861
+ }
862
+
741
863
  // ── Conversation API ────────────────────────────────────────────────────
742
864
 
743
865
  /** Begin a new conversation, optionally restoring from a previously-saved
@@ -5,7 +5,7 @@
5
5
  // a list of PipelineMechanism objects — it never imports a mechanism-specific
6
6
  // type and never has a special-case branch for any mechanism.
7
7
  //
8
- // The four constraints of the free-will architecture (§14.5):
8
+ // The four constraints of the free-will architecture (mechanism-market.md):
9
9
  // 1. DECOUPLING — mechanisms import nothing from each other or from pipeline.
10
10
  // 2. DECLARED COMPETENCE — floor() returns null when impossible, a number when
11
11
  // possible. Binary, auditable, no learned scores.
@@ -148,17 +148,17 @@ export class Precomputed {
148
148
  );
149
149
  }
150
150
 
151
- // REMOVED — the WIDE exhaustive-√N resonance list (`wideResonance`). It ran
151
+ // REMOVED — the WIDE exhaustive-√N resonance list (`wideResonance`). It ran
152
152
  // `resonate(guide, √N, exhaustive=true)` whenever the top hit cleared
153
- // conceptThreshold, so consumers could look "past the top-k". Every consumer
153
+ // conceptThreshold, so consumers could look "past the top-k". Every consumer
154
154
  // only ever needed ≤ 2·recallQueryK proposals (the substitution bridge's own
155
155
  // candidate cap) or a content-addressed answer (prefix completion's
156
- // formsOpenedBy), and every proposal is byte-verified downstream (§2.3), so
157
- // the exhaustive scan bought recall at O(index) cost for an O(k) need —
158
- // measured: 244K annVectorReads per refusing query, ~1.5 s, every answer
159
- // byte-identical to a top-k read. The two consumers now read `resonance()`
160
- // (the one top-k read) and the write side's window index respectively — see
161
- // recall.ts and prefix-completion.ts.
156
+ // formsOpenedBy), and every proposal is byte-verified downstream
157
+ // (exact-vs-approximate.md), so the exhaustive scan bought recall at O(index)
158
+ // cost for an O(k) need — measured: 244K annVectorReads per refusing query,
159
+ // ~1.5 s, every answer byte-identical to a top-k read. The two consumers now
160
+ // read `resonance()` (the one top-k read) and the write side's window index
161
+ // respectively — see recall.ts and prefix-completion.ts.
162
162
 
163
163
  private _frames?: Promise<ReadonlyArray<FrameInstance>>;
164
164
  /** THE FRAME INVENTORY — every ranked candidate that reads as an instance of
@@ -166,14 +166,16 @@ export class Precomputed {
166
166
  * ({@link FrameInstance}). The one place the engine represents "a position
167
167
  * whose occupant comes from the context rather than the corpus".
168
168
  *
169
- * AN INVENTORY, NOT AN ELECTION. It reports every pairing and elects no
170
- * frame, deliberately: a slot is a property of a PAIRING, not of the query,
171
- * and different candidates put slots in different places. Committing to one
172
- * reading here would push whichever consumer asked first onto everyone else
173
- * — the market's decoupling (§2.6) broken from inside the shared container,
174
- * and the population error §2.7 names. Each consumer groups and commits
175
- * for its own question; reference elects the modal slot signature, and a
176
- * consumer wanting a different reading is not fighting this one.
169
+ * AN INVENTORY, NOT AN ELECTION. It reports every pairing and elects no
170
+ * frame,
171
+ * deliberately: a slot is a property of a PAIRING, not of the query, and
172
+ * different candidates put slots in different places. Committing to one
173
+ * reading here would push whichever consumer asked first onto everyone else —
174
+ * the market's decoupling (mechanism-market.md) broken from inside the shared
175
+ * container, and the population error commonality.md names. Each consumer
176
+ * groups and commits for its own question; reference elects the modal slot
177
+ * signature, and a consumer wanting a different reading is not fighting this
178
+ * one.
177
179
  *
178
180
  * NO LICENCE EITHER. Knowing a span is variable is safe for every consumer
179
181
  * — it can only improve an alignment. Knowing one may be VOICED through is
@@ -189,9 +191,10 @@ export class Precomputed {
189
191
  const capBytes = this.query.length * W;
190
192
  const out: FrameInstance[] = [];
191
193
  for (const h of await this.resonance()) {
192
- // REJECT BY LENGTH BEFORE RECONSTRUCTING (§2.8): `contentLen` is an
193
- // indexed read, `bytesPrefix` rebuilds a subtree. ONLY the phrase-scale
194
- // cap is applied — it is a bounded-read discipline, not a judgement.
194
+ // REJECT BY LENGTH BEFORE RECONSTRUCTING (bounded-reads.md):
195
+ // `contentLen` is an indexed read, `bytesPrefix` rebuilds a subtree.
196
+ // ONLY the phrase-scale cap is applied — it is a bounded-read
197
+ // discipline, not a judgement.
195
198
  //
196
199
  // A LOWER bound was here too (`dominates(len, query.length)`, on the
197
200
  // reasoning that a candidate shorter than half the query cannot supply
@@ -519,7 +522,8 @@ function computeWeave(
519
522
  // IDF — gates the aligner has no equivalent of.
520
523
  //
521
524
  // So the climb PROPOSES the pairing (which structure, which query span) and
522
- // bytes DECIDE its terms (§2.3). Three gates, each one measured:
525
+ // bytes DECIDE its terms (exact-vs-approximate.md). Three gates, each one
526
+ // measured:
523
527
  //
524
528
  // • it may only take query bytes NO literal run claimed. Run inline with
525
529
  // phase 1 this did the opposite of "exact decides" — a higher-ranked
@@ -130,15 +130,15 @@ export interface NarrowDecisionData {
130
130
  }
131
131
 
132
132
  /** Structured payload of the "regimePrediction" rationale step — the R8
133
- * observation exposed as data. After the first mechanism (cover, which §2.6
134
- * runs first) grounds or abstains, the market's whole outcome is already
135
- * determined by the one cost ladder: the consensus climb runs exactly when
136
- * `worthRunning(2 * STEP)` is true — CAST (floor 2·STEP) is the cheapest
137
- * mechanism that first-touches it, and confluence (3·STEP) / extraction
138
- * (CONCEPT+STEP) are only reached after CAST is. An incumbent at or below
139
- * that floor prunes CAST and, with it, the climb (retrieval); anything above
140
- * — or no incumbent — runs the full market and the climb (composition).
141
- * Purely observational; never read by inference. */
133
+ * observation exposed as data. After the first mechanism (cover, which
134
+ * mechanism-market.md runs first) grounds or abstains, the market's whole
135
+ * outcome is already determined by the one cost ladder: the consensus climb
136
+ * runs exactly when `worthRunning(2 * STEP)` is true — CAST (floor 2·STEP) is
137
+ * the cheapest mechanism that first-touches it, and confluence (3·STEP) /
138
+ * extraction (CONCEPT+STEP) are only reached after CAST is. An incumbent at or
139
+ * below that floor prunes CAST and, with it, the climb (retrieval); anything
140
+ * above — or no incumbent — runs the full market and the climb (composition).
141
+ * Purely observational; never read by inference. */
142
142
  export interface RegimePredictionData {
143
143
  version: 1;
144
144
  /** retrieval | composition — the two regimes R1 measured as a ~100× cost
@@ -187,13 +187,13 @@ export async function think(
187
187
  // ── Pre-computation ──────────────────────────────────────────────────
188
188
  const mechanisms = mechs ?? defaultMechanisms;
189
189
  const meter = ctx.meter;
190
- // recognition is a shared analysis (§2.14 contract 5): it does the query's
190
+ // recognition is a shared analysis (meter.md contract 5): it does the query's
191
191
  // own store work (perceive → foldTree → resolve), which used to land in
192
192
  // `think` and in nothing narrower — the meter's one accounting surface must
193
193
  // charge it to itself, exactly as attention/weave/resonance are charged.
194
- // SYNCHRONOUS phase: recognition is on the sync side of §2.10's seam, so it
195
- // is timed with `timeSync` — wrapping it in a promise would make a profiled
196
- // response await where an unprofiled one does not.
194
+ // SYNCHRONOUS phase: recognition is on the sync side of meter.md's seam, so
195
+ // it is timed with `timeSync` — wrapping it in a promise would make a
196
+ // profiled response await where an unprofiled one does not.
197
197
  const rec = meter
198
198
  ? meter.timeSync("recognise", () => recognise(ctx, query))
199
199
  : recognise(ctx, query);
@@ -225,15 +225,15 @@ export async function think(
225
225
  }
226
226
  }
227
227
 
228
- // Phase 2: the shared pre-computation container. Eager fields only
229
- // (recognition, computed spans, guide) — every expensive analysis
230
- // (consensus climb, weave, span-shape classification) is a lazily-cached
231
- // method on Precomputed, first-touched by whichever mechanism's floor
232
- // survives its cheap gates and the worthRunning check. A query no
233
- // mechanism climbs for (e.g. one an extension decided) never climbs.
234
- // NOT phased: the constructor itself is trivial (it only derives `k`), so a
235
- // phase here would add a zero-work entry to every profiled report — the meter
236
- // attributes WORK (§2.14); the trace already represents structure.
228
+ // Phase 2: the shared pre-computation container. Eager fields only
229
+ // (recognition, computed spans, guide) — every expensive analysis (consensus
230
+ // climb, weave, span-shape classification) is a lazily-cached method on
231
+ // Precomputed, first-touched by whichever mechanism's floor survives its
232
+ // cheap gates and the worthRunning check. A query no mechanism climbs for
233
+ // (e.g. one an extension decided) never climbs. NOT phased: the constructor
234
+ // itself is trivial (it only derives `k`), so a phase here would add a
235
+ // zero-work entry to every profiled report — the meter attributes WORK
236
+ // (meter.md); the trace already represents structure.
237
237
  const pre = new Precomputed(ctx, query, rec, computed, ctx._edgeGuide);
238
238
 
239
239
  // ── Grounding: ONE lightest-derivation choice among the mechanisms ────
@@ -298,16 +298,17 @@ export async function think(
298
298
  const worthRunning = (floor: number) =>
299
299
  best === null || grade(floor) < grade(best.weight);
300
300
 
301
- // REGIME PREDICTION (R8) — observational only. Once the FIRST mechanism has
302
- // had its turn (cover, which §2.6 places first and floors at 0), the market's
303
- // outcome is already determined by the one cost ladder: the consensus climb
304
- // runs exactly when `worthRunning(2 * STEP)` is true — CAST (floor 2·STEP) is
305
- // the cheapest mechanism that first-touches it, so an incumbent at or below
306
- // grade 2 prunes CAST and, with it, confluence (3·STEP) and extraction
307
- // (CONCEPT+STEP) (retrieval); anything above — or no incumbent — runs the
308
- // full market and the climb (composition). The predicate is `worthRunning`,
309
- // the same function the loop itself uses — nothing is computed here that the
310
- // engine had not already computed, and nothing is read back by inference.
301
+ // REGIME PREDICTION (R8) — observational only. Once the FIRST mechanism has
302
+ // had its turn (cover, which mechanism-market.md places first and floors at
303
+ // 0), the market's outcome is already determined by the one cost ladder: the
304
+ // consensus climb runs exactly when `worthRunning(2 * STEP)` is true — CAST
305
+ // (floor 2·STEP) is the cheapest mechanism that first-touches it, so an
306
+ // incumbent at or below grade 2 prunes CAST and, with it, confluence (3·STEP)
307
+ // and extraction (CONCEPT+STEP) (retrieval); anything above — or no incumbent
308
+ // — runs the full market and the climb (composition). The predicate is
309
+ // `worthRunning`, the same function the loop itself uses — nothing is
310
+ // computed here that the engine had not already computed, and nothing is read
311
+ // back by inference.
311
312
  //
312
313
  // EMITTED BEFORE THE SECOND MECHANISM'S FLOOR, never after some mechanism's
313
314
  // run: a "prediction" published after the fact could assert "the climb will
@@ -544,12 +545,40 @@ export async function think(
544
545
  ctx.store.nextFirst(id, hubBound(ctx)).map((n) => read(ctx, n))
545
546
  )
546
547
  : [];
548
+ // REPORTABLE, NOT SILENT. A declared-complete grounding ends the derivation
549
+ // here, and that decision is part of the derivation's shape: the reader of a
550
+ // rationale must be able to see that the chain stopped because the mechanism
551
+ // claimed the query WAS the context, not because nothing followed. The step
552
+ // carries the claim, not a re-description of the answer — the extension is
553
+ // skipped, so there is no output item to show.
554
+ if (decided.complete) {
555
+ ctx.trace?.step(
556
+ "completeGrounding",
557
+ [rItem(answer, provenance)],
558
+ [],
559
+ "grounding declared complete — the query IS the context, so " +
560
+ "post-grounding extension is skipped",
561
+ );
562
+ }
563
+ // THE REASONER JUDGES ITS OWN EXTENSIONS BY THE PIPELINE'S REMAINDER, not by
564
+ // the ladder's `accounted` — and by the SAME reading the fuse gate below uses,
565
+ // with the same W floor. `accounted` is a COST quantity (measured: a query
566
+ // fully explained by one computed span plus bridged connectors reports
567
+ // `accounted: []` while nothing is unexplained), and a remainder under one
568
+ // river-fold quantum is bridging punctuation, never a second topic — so it
569
+ // licenses no extension and blocks none.
570
+ const explained: Array<[number, number]> = [
571
+ ...decided.accounted,
572
+ ...pre.computed.map((u): [number, number] => [u.i, u.j]),
573
+ ];
574
+ const uncovered = unexplainedSpans(query.length, explained)
575
+ .filter(([a, b]) => b - a >= ctx.space.maxGroup);
547
576
  const reasoned = decided.complete ? answer : meter
548
577
  ? await meter.time(
549
578
  "reason",
550
- () => reason(ctx, query, answer, preConsumed, pre, voiced),
579
+ () => reason(ctx, query, answer, preConsumed, pre, voiced, uncovered),
551
580
  )
552
- : await reason(ctx, query, answer, preConsumed, pre, voiced);
581
+ : await reason(ctx, query, answer, preConsumed, pre, voiced, uncovered);
553
582
 
554
583
  // Fuse only when the query has a genuine REMAINDER no mechanism's
555
584
  // structural evidence touched at all. `decided.accounted` alone
@@ -567,10 +596,6 @@ export async function think(
567
596
  // observed: a single space between two fully-computed arithmetic spans
568
597
  // ("2+2 3+3") registered as "unaccounted" and pulled in an unrelated
569
598
  // corpus fact, corrupting "4 6" into "4 63".
570
- const explained: Array<[number, number]> = [
571
- ...decided.accounted,
572
- ...pre.computed.map((u): [number, number] => [u.i, u.j]),
573
- ];
574
599
  const remainder = unaccounted(explained);
575
600
  // Whether the winning candidate's entire recognised substance is
576
601
  // COMPUTED — every accounted span exactly a pre.computed span, nothing
@@ -59,11 +59,11 @@ export function perceiveKey(
59
59
  /** Perceive input into a content-defined tree (the river fold).
60
60
  * Deterministic — identical bytes always produce an identical tree.
61
61
  *
62
- * `boundaries` is an optional sorted list of proper byte offsets where the
63
- * fold must split so that each prefix segment folds identically to how it
64
- * folded when it was learned (§10.3 stable-prefix contract). Only the
65
- * CALLER — who assembled the multi-turn context — knows where those
66
- * boundaries are; the geometry never guesses them from the bytes. */
62
+ * `boundaries` is an optional sorted list of proper byte offsets where the fold
63
+ * must split so that each prefix segment folds identically to how it folded
64
+ * when it was learned (fold-contract.md stable-prefix contract). Only the
65
+ * CALLER — who assembled the multi-turn context — knows where those boundaries
66
+ * are; the geometry never guesses them from the bytes. */
67
67
  export function perceive(
68
68
  ctx: MindContext,
69
69
  input: Input,
@@ -43,6 +43,10 @@ export async function reason(
43
43
  preConsumed: ReadonlySet<number>,
44
44
  pre: Precomputed,
45
45
  voiced: readonly Uint8Array[] = [],
46
+ /** The query material the GROUNDING left uncovered — the cost ladder's own
47
+ * `unaccounted` spans. Only the reasoner's OWN extensions are judged
48
+ * against it; a mechanism carrying its own `used` set owns its shape. */
49
+ uncovered: readonly (readonly [number, number])[] = [],
46
50
  ): Promise<Uint8Array> {
47
51
  // Echo guard: a query that is ITSELF a learnt continuation (some context's
48
52
  // answer) is being asked back at the system — hopping forward from it would
@@ -200,6 +204,57 @@ export async function reason(
200
204
  const fc = await follow(ctx, pivot, qv);
201
205
  consumeAll(pivot);
202
206
  if (fc === null || bytesEqual(fc, cur) || restatesQuery(query, fc)) break;
207
+ // WHOSE EXTENSION IS THIS?
208
+ //
209
+ // `voiced` is what the mechanism WITHHELD (the pipeline sends the used
210
+ // anchors' CONTINUATIONS, not their bytes — see pipeline's own note), so a
211
+ // non-empty `voiced` means exactly what that note says: the grounding came
212
+ // from a mechanism that carries its own short `used` set (cast/join) and
213
+ // therefore owns the shape of its answer. The further terms inside such a
214
+ // seat are legitimately followable — test/29 C3's `Mona Lisa` lives inside
215
+ // the voiced seat and leads on to a fact about neither analog.
216
+ //
217
+ // Every other grounding is ordinary, and an extension of it is the
218
+ // reasoner's own inference: it is taken only while question material the
219
+ // grounding left uncovered remains AND the step carries some of it, judged
220
+ // by the mind's own line between chance and evidence — one W-byte window,
221
+ // no word notion, no character class, no threshold. Measured: the drift's
222
+ // second step (`the Eiffel Tower is in Paris` after `Paris is famous for
223
+ // the Eiffel Tower`) carries no window of `" famous for"` and is refused,
224
+ // while the first carries it. Terminates by a real argument: the uncovered
225
+ // material is finite and each taken extension must carry some of it.
226
+ const producerOwnsShape = voiced.length > 0;
227
+ if (!producerOwnsShape && uncovered.length > 0) {
228
+ const W = ctx.space.maxGroup;
229
+ let progress = false;
230
+ for (const [a, b] of uncovered) {
231
+ for (let i = a; i + W <= b && !progress; i++) {
232
+ if (indexOf(fc, query.subarray(i, i + W), 0) >= 0) progress = true;
233
+ }
234
+ if (progress) break;
235
+ }
236
+ if (!progress) {
237
+ // THE BRAKE, MADE VISIBLE. The reasoner declines a step that carries
238
+ // none of the material the grounding left uncovered — the drift the
239
+ // extension tests pin. A refusal that leaves no trace is the kind of
240
+ // silent cut AGENTS §6 forbids: the rationale is where a reader learns
241
+ // that an extension was declined for want of question material, and
242
+ // where the next person sees why the chain stopped here. Measured with
243
+ // the check disabled, test/110 and test/116 fail — so this brake is the
244
+ // only thing keeping the extension honest until the pivot reports its
245
+ // own accounted spans and the ladder can judge it instead.
246
+ const left = uncovered.reduce((n, [a, b]) => n + (b - a), 0);
247
+ ctx.trace?.step(
248
+ "pivotRefused",
249
+ [rItem(cur, "answer"), rItem(query, "query")],
250
+ uncovered.map(([a, b]) => rItem(query.subarray(a, b), "uncovered")),
251
+ `the step carries none of the question material the grounding left ` +
252
+ `uncovered (${left} byte(s) in ${uncovered.length} span(s)) — refused`,
253
+ );
254
+ break;
255
+ }
256
+ }
257
+ if (ctx.meter) ctx.meter.pivotSteps++;
203
258
  t ??= ctx.trace?.enter("reason", [rItem(startedFrom, "grounded")]);
204
259
  ctx.trace?.step(
205
260
  "pivotStep",
@@ -34,19 +34,20 @@ import type { Leaf, Site } from "./graph-search.js";
34
34
  *
35
35
  * Both O(n · maxGroup) bounded O(1) probes — never a scan of the corpus.
36
36
  *
37
- * ONE READING PER BYTE STREAM, deliberately: there is no "cheap mode" that
38
- * skips the edge-trim fallbacks. A `trimmed` variant was tried and REFUTED
39
- * twice over. Its premise — "the trims only recover misaligned FRAGMENTS, so
40
- * a consumer whose gate rejects fragments loses nothing" — is false: the
41
- * left/right trim loops below exist precisely to find WHOLE trained forms
42
- * embedded at an offset the query's own fold did not cut, and such a form has
43
- * no structural parents or containers, so it passes the pivot's fragment gate
44
- * and is exactly the candidate a multi-hop chain steps through. Skipping them
45
- * narrows the pivot's evidence silently. And a per-caller variant has to key
46
- * the memo by the variant, which breaks the "computed at most once" property
47
- * (§2.11): the pipeline recognises a grounded answer untrimmed for
48
- * `preConsumed`, and the pivot then recognises the same bytes again — the
49
- * saving inverts into a doubling on the path it was measured for. */
37
+ * ONE READING PER BYTE STREAM, deliberately: there is no "cheap mode" that
38
+ * skips
39
+ * the edge-trim fallbacks. A `trimmed` variant was tried and REFUTED twice
40
+ * over. Its premise — "the trims only recover misaligned FRAGMENTS, so a
41
+ * consumer whose gate rejects fragments loses nothing" — is false: the
42
+ * left/right trim loops below exist precisely to find WHOLE trained forms
43
+ * embedded at an offset the query's own fold did not cut, and such a form has
44
+ * no structural parents or containers, so it passes the pivot's fragment gate
45
+ * and is exactly the candidate a multi-hop chain steps through. Skipping them
46
+ * narrows the pivot's evidence silently. And a per-caller variant has to key
47
+ * the memo by the variant, which breaks the "computed at most once" property
48
+ * (memoization.md): the pipeline recognises a grounded answer untrimmed for
49
+ * `preConsumed`, and the pivot then recognises the same bytes again — the
50
+ * saving inverts into a doubling on the path it was measured for. */
50
51
  export function recognise(
51
52
  ctx: MindContext,
52
53
  bytes: Uint8Array,
@@ -180,16 +181,15 @@ function recogniseImpl(ctx: MindContext, bytes: Uint8Array): Recognition {
180
181
  // and read the answer from `starts`, which is exactly {0, W, 2W, …}
181
182
  // because riverFold groups fixed-arity — arithmetic, not evidence.
182
183
  //
183
- // Measured on the 17.9M-node store, over the sites of 7 probes (1 good,
184
- // 11 junk by hand-labelling, corrected for whole-query forms):
185
- // len >= W rejects "hi"(2) "of"(2) "is"(2) "di"(2) "the"(3),
186
- // admits "Eiffel Tower"(12) and both whole-query forms
187
- // len >= W-1 admits "the" — W-1 is the write side's straddle
188
- // neighbour for RETRIEVAL, never a claim about units
189
- // §2.7 saturation admits 11/11 junk: edgeAncestors on a site node
190
- // reaches 1..48 contexts, so dominates(ctx, N) needs
191
- // ctx > 162805 and never fires; every site reads DISC
192
- // rarity does not separate: "hi" has 1 container, "the" 572
184
+ // Measured on the 17.9M-node store, over the sites of 7 probes (1 good, 11
185
+ // junk by hand-labelling, corrected for whole-query forms): len >= W
186
+ // rejects "hi"(2) "of"(2) "is"(2) "di"(2) "the"(3), admits "Eiffel
187
+ // Tower"(12) and both whole-query forms len >= W-1 admits "the" — W-1 is
188
+ // the write side's straddle neighbour for RETRIEVAL, never a claim about
189
+ // units commonality.md saturation admits 11/11 junk: edgeAncestors on a
190
+ // site node reaches 1..48 contexts, so dominates(ctx, N) needs ctx > 162805
191
+ // and never fires; every site reads DISC rarity does not separate: "hi" has
192
+ // 1 container, "the" 572
193
193
  //
194
194
  // A span covering the WHOLE query is exempt: then it is not a fragment of
195
195
  // something longer, it is the question ("hi" asked on its own).