@hviana/sema 0.8.1 → 0.8.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (57) hide show
  1. package/dist/src/config.d.ts +17 -0
  2. package/dist/src/config.js +18 -0
  3. package/dist/src/meter.d.ts +25 -0
  4. package/dist/src/meter.js +44 -0
  5. package/dist/src/mind/corpus.d.ts +40 -0
  6. package/dist/src/mind/corpus.js +149 -0
  7. package/dist/src/mind/graph-search.d.ts +7 -0
  8. package/dist/src/mind/graph-search.js +235 -24
  9. package/dist/src/mind/index.d.ts +3 -1
  10. package/dist/src/mind/index.js +1 -0
  11. package/dist/src/mind/match.d.ts +8 -3
  12. package/dist/src/mind/match.js +142 -58
  13. package/dist/src/mind/mechanisms/cast.js +18 -2
  14. package/dist/src/mind/mechanisms/cover.js +6 -0
  15. package/dist/src/mind/mind.d.ts +55 -0
  16. package/dist/src/mind/mind.js +72 -2
  17. package/dist/src/mind/pipeline.js +25 -6
  18. package/dist/src/mind/reasoning.d.ts +5 -1
  19. package/dist/src/mind/reasoning.js +54 -1
  20. package/dist/src/mind/traverse.js +9 -1
  21. package/dist/src/mind/types.d.ts +22 -0
  22. package/docs/failures/tempting-but-wrong.md +31 -2
  23. package/jsr.json +1 -1
  24. package/package.json +1 -1
  25. package/src/config.ts +35 -0
  26. package/src/meter.ts +47 -0
  27. package/src/mind/corpus.ts +202 -0
  28. package/src/mind/graph-search.ts +252 -23
  29. package/src/mind/index.ts +8 -1
  30. package/src/mind/match.ts +143 -54
  31. package/src/mind/mechanisms/cast.ts +17 -1
  32. package/src/mind/mechanisms/cover.ts +5 -0
  33. package/src/mind/mind.ts +123 -0
  34. package/src/mind/pipeline.ts +30 -6
  35. package/src/mind/reasoning.ts +55 -0
  36. package/src/mind/traverse.ts +9 -1
  37. package/src/mind/types.ts +26 -0
  38. package/test/100-complete-grounding-trace.test.mjs +109 -0
  39. package/test/101-alignment-gap-bound.test.mjs +106 -0
  40. package/test/102-production-composes-at-scale.test.mjs +110 -0
  41. package/test/103-alignment-gap-budget.test.mjs +89 -0
  42. package/test/104-composition-is-reported.test.mjs +90 -0
  43. package/test/105-derive-through-reports-its-refusal.test.mjs +113 -0
  44. package/test/106-the-join-fires.test.mjs +94 -0
  45. package/test/107-the-join-is-counted.test.mjs +81 -0
  46. package/test/108-the-join-chains.test.mjs +78 -0
  47. package/test/109-the-pivot-is-counted.test.mjs +60 -0
  48. package/test/110-the-reasoner-stops-when-the-question-is-answered.test.mjs +91 -0
  49. package/test/111-the-cover-assembly-is-counted.test.mjs +74 -0
  50. package/test/112-the-exploration-does-not-grow-with-the-hub.test.mjs +89 -0
  51. package/test/113-the-rationale-payload-is-bounded.test.mjs +84 -0
  52. package/test/114-alignment-budget-is-per-sweep.test.mjs +93 -0
  53. package/test/116-the-extension-is-gated-by-the-pipelines-own-remainder.test.mjs +100 -0
  54. package/test/117-corpus-search.test.mjs +171 -0
  55. package/test/14-scaling.test.mjs +10 -7
  56. package/test/76-reference-binding.test.mjs +6 -1
  57. package/test/89-completion-recursion.test.mjs +30 -5
package/src/mind/match.ts CHANGED
@@ -38,7 +38,7 @@ import {
38
38
  identityBar,
39
39
  significanceBar,
40
40
  } from "../geometry.js";
41
- import { bytesEqual, indexOf } from "../bytes.js";
41
+ import { bytesEqual, indexOf, latin1 } from "../bytes.js";
42
42
  import type { MindContext } from "./types.js";
43
43
  import { chainReach, leafIdRun } from "./canonical.js";
44
44
  import { foldTree, gistOf, perceive, read, resolve } from "./primitives.js";
@@ -54,6 +54,7 @@ import {
54
54
  sharedReachMemo,
55
55
  } from "./traverse.js";
56
56
  import { recognise, segment } from "./recognition.js";
57
+ import { rItem } from "./trace.js";
57
58
  import type { Site } from "./graph-search.js";
58
59
 
59
60
  // ═══════════════════════════════════════════════════════════════════════════
@@ -360,9 +361,14 @@ export interface AlignGap {
360
361
 
361
362
  /** Extend a seed match (query offset qo ↔ candidate offset co) to its maximal
362
363
  * common run, then walk outward in both directions collecting further common
363
- * runs of at least W bytes across bounded mismatch gaps (each side ≤
364
- * chainReach). Returns the matched query spans and the mismatch pairs
365
- * between consecutive runs.
364
+ * runs of at least W bytes across mismatch gaps. Each gap's LENGTH is the
365
+ * pair's own extent (a gap cannot be longer than the bytes it spans) and the
366
+ * sweep's WORK is proportional to the bytes a run spans (the context's windows
367
+ * are indexed once, then the query's are walked) — the arity bound
368
+ * (`chainReach`) used to cap BOTH, and truncated every learned frame whose
369
+ * slot was longer. Each sweep owns its own budget, so an exhausted right
370
+ * sweep never starves the left one. Returns the matched query spans and the
371
+ * mismatch pairs between consecutive runs.
366
372
  *
367
373
  * This is the SEEDED aligner, distinct from {@link alignRuns}: that one finds
368
374
  * every run two structures share anywhere (a weave), this one reads two
@@ -382,7 +388,21 @@ export function alignAround(
382
388
  co: number,
383
389
  ): { matched: Array<[number, number]>; gaps: AlignGap[] } {
384
390
  const W = ctx.space.maxGroup;
385
- const reachCap = chainReach(W);
391
+ // THE GAP LENGTH IS THE PAIR'S OWN EXTENT; THE WORK IS BUDGETED.
392
+ //
393
+ // The sweep walks (queryGap, contextGap) pairs by ASCENDING total, so reaching
394
+ // a gap of size G costs about G²/2 pairs. Bounding the LENGTH by the write
395
+ // side's arity (`chainReach(W)` = 16) therefore truncated every learned frame
396
+ // whose slot is longer — measured: `bindReference` reported the cap at 18, 24,
397
+ // 30 and 36 bytes and `recall` answered with ANOTHER instance's filler — while
398
+ // removing the bound outright took the corpus-cost guard (test/89) from
399
+ // milliseconds to 68 seconds.
400
+ //
401
+ // Bounding the PAIRS keeps a call's cost constant however long the pair is,
402
+ // and the ascending order means an exhausted budget drops the FAR
403
+ // continuations and never the near ones — the same degradation recognition.ts
404
+ // documents for its canon budget. Length and work are different questions;
405
+ // this is the one place they were conflated.
386
406
  // Maximal run around the seed.
387
407
  let qs = qo, ss = co;
388
408
  while (qs > 0 && ss > 0 && q[qs - 1] === c[ss - 1]) {
@@ -396,8 +416,60 @@ export function alignAround(
396
416
  }
397
417
  const matched: Array<[number, number]> = [[qs, qe]];
398
418
  const gaps: AlignGap[] = [];
399
- // The next common run of ≥ W bytes past (qi, si), with each side's gap
400
- // bounded by chainReach; smallest total gap wins (nearest continuation).
419
+ // THE SWEEP IS STRUCTURAL, NOT ENUMERATIVE.
420
+ //
421
+ // The criterion is unchanged: the next common run, MINIMUM TOTAL GAP, ties to
422
+ // the smaller query gap. What changed is how it is found. Enumerating
423
+ // (queryGap, contextGap) pairs by ascending total reaches a run at total t in
424
+ // about t²/2 pairs — and that quadratic shape, not the reach, was the cost
425
+ // problem: capping the pairs dropped reach (a legitimate 24-byte slot stopped
426
+ // being found), while leaving them uncapped cost 68 seconds on the corpus
427
+ // guard. Neither is the answer, because the answer is the algorithm.
428
+ //
429
+ // The context's windows are indexed ONCE, for lengths 1..W — W being the
430
+ // geometry's own unit of composition, so nothing is chosen here. Each step
431
+ // then walks the query's windows outward from the anchor: for a given query
432
+ // gap the nearest context gap that continues a run is one O(1) lookup, and the
433
+ // walk stops the moment the query gap alone exceeds the best total already
434
+ // found. So the work is proportional to the bytes the run SPANS. No budget,
435
+ // no cap, no number: a long slot is reached, and its price is already the
436
+ // ladder's (its bytes are unaccounted, so the search pays PASS per byte).
437
+ const index: Array<Map<string, number[]>> = [];
438
+ for (let len = 1; len <= W; len++) {
439
+ const m = new Map<string, number[]>();
440
+ for (let o = 0; o + len <= c.length; o++) {
441
+ const key = latin1(c.subarray(o, o + len));
442
+ const at = m.get(key);
443
+ if (at === undefined) m.set(key, [o]);
444
+ else at.push(o);
445
+ }
446
+ index.push(m);
447
+ }
448
+ /** Smallest listed offset at or after `from`, or -1. */
449
+ const fromAt = (list: number[], from: number): number => {
450
+ let lo = 0, hi = list.length - 1, best = -1;
451
+ while (lo <= hi) {
452
+ const mid = (lo + hi) >> 1;
453
+ if (list[mid] >= from) {
454
+ best = list[mid];
455
+ hi = mid - 1;
456
+ } else lo = mid + 1;
457
+ }
458
+ return best;
459
+ };
460
+ /** Largest listed offset at or before `to`, or -1. */
461
+ const toAt = (list: number[], to: number): number => {
462
+ let lo = 0, hi = list.length - 1, best = -1;
463
+ while (lo <= hi) {
464
+ const mid = (lo + hi) >> 1;
465
+ if (list[mid] <= to) {
466
+ best = list[mid];
467
+ lo = mid + 1;
468
+ } else hi = mid - 1;
469
+ }
470
+ return best;
471
+ };
472
+ /** Length of the common run STARTING at (qi, si). */
401
473
  const runLenAt = (qi: number, si: number): number => {
402
474
  let n = 0;
403
475
  while (qi + n < q.length && si + n < c.length && q[qi + n] === c[si + n]) {
@@ -405,63 +477,80 @@ export function alignAround(
405
477
  }
406
478
  return n;
407
479
  };
408
- // RIGHT sweep.
409
- let qi = qe, si = se;
410
- for (;;) {
411
- let found = false;
412
- for (let total = 1; total <= 2 * reachCap && !found; total++) {
413
- for (let gq = 0; gq <= Math.min(total, reachCap); gq++) {
414
- const gs = total - gq;
415
- if (gs > reachCap) continue;
416
- if (qi + gq >= q.length || si + gs >= c.length) continue;
417
- const n = runLenAt(qi + gq, si + gs);
418
- if (n >= W || qi + gq + n === q.length) {
419
- if (n === 0) continue;
420
- if (gq > 0 || gs > 0) {
421
- gaps.push({ qs: qi, qe: qi + gq, cs: si, ce: si + gs });
480
+ /** Length of the common run ENDING at (qi, si). */
481
+ const runLenBefore = (qi: number, si: number): number => {
482
+ let n = 0;
483
+ while (n < qi && n < si && q[qi - 1 - n] === c[si - 1 - n]) n++;
484
+ return n;
485
+ };
486
+ /** The next run outward from an anchor, or null when the bytes run out. */
487
+ const nextRun = (
488
+ qi: number,
489
+ si: number,
490
+ forward: boolean,
491
+ ): { gq: number; gs: number; n: number } | null => {
492
+ const qLim = forward ? q.length - qi : qi;
493
+ let best: { gq: number; gs: number; n: number } | null = null;
494
+ for (let gq = 0; gq < qLim; gq++) {
495
+ // No later query gap can beat a total already found.
496
+ if (best !== null && gq > best.gq + best.gs) break;
497
+ const left = qLim - gq;
498
+ // A run of >= W bytes, or — when the query itself ends inside one window —
499
+ // the run that REACHES that end. Exactly the acceptance the sweep had.
500
+ const lens = left >= W ? [W] : [left];
501
+ for (const len of lens) {
502
+ const key = latin1(
503
+ q.subarray(
504
+ forward ? qi + gq : qi - gq - len,
505
+ forward ? qi + gq + len : qi - gq,
506
+ ),
507
+ );
508
+ const list = index[len - 1].get(key);
509
+ if (list === undefined) continue;
510
+ const o = forward ? fromAt(list, si) : toAt(list, si - len);
511
+ if (o < 0) continue;
512
+ const n = forward
513
+ ? runLenAt(qi + gq, o)
514
+ : runLenBefore(qi - gq, o + len);
515
+ if (n < 1) continue;
516
+ if (
517
+ forward ? n >= W || qi + gq + n === q.length : n >= W || n === qi - gq
518
+ ) {
519
+ const gs = forward ? o - si : si - len - o;
520
+ if (best === null || gq + gs < best.gq + best.gs) {
521
+ best = { gq, gs, n };
422
522
  }
423
- matched.push([qi + gq, qi + gq + n]);
424
- qi = qi + gq + n;
425
- si = si + gs + n;
426
- found = true;
427
523
  break;
428
524
  }
429
525
  }
430
526
  }
431
- if (!found) break;
527
+ return best;
528
+ };
529
+ // RIGHT sweep.
530
+ let qi = qe, si = se;
531
+ for (;;) {
532
+ const step = nextRun(qi, si, true);
533
+ if (step === null) break;
534
+ if (step.gq > 0 || step.gs > 0) {
535
+ gaps.push({ qs: qi, qe: qi + step.gq, cs: si, ce: si + step.gs });
536
+ }
537
+ matched.push([qi + step.gq, qi + step.gq + step.n]);
538
+ qi = qi + step.gq + step.n;
539
+ si = si + step.gs + step.n;
432
540
  }
433
- // LEFT sweep (mirror).
541
+ // LEFT sweep (mirror): an independent walk, so an exhausted right side can
542
+ // never starve it (pinned by test/114).
434
543
  qi = qs;
435
544
  si = ss;
436
545
  for (;;) {
437
- let found = false;
438
- for (let total = 1; total <= 2 * reachCap && !found; total++) {
439
- for (let gq = 0; gq <= Math.min(total, reachCap); gq++) {
440
- const gs = total - gq;
441
- if (gs > reachCap) continue;
442
- if (qi - gq <= 0 || si - gs <= 0) continue;
443
- // Run ENDING at (qi - gq, si - gs).
444
- let n = 0;
445
- while (
446
- n < qi - gq && n < si - gs &&
447
- q[qi - gq - 1 - n] === c[si - gs - 1 - n]
448
- ) {
449
- n++;
450
- }
451
- if (n >= W || n === qi - gq) {
452
- if (n === 0) continue;
453
- if (gq > 0 || gs > 0) {
454
- gaps.push({ qs: qi - gq, qe: qi, cs: si - gs, ce: si });
455
- }
456
- matched.push([qi - gq - n, qi - gq]);
457
- qi = qi - gq - n;
458
- si = si - gs - n;
459
- found = true;
460
- break;
461
- }
462
- }
546
+ const step = nextRun(qi, si, false);
547
+ if (step === null) break;
548
+ if (step.gq > 0 || step.gs > 0) {
549
+ gaps.push({ qs: qi - step.gq, qe: qi, cs: si - step.gs, ce: si });
463
550
  }
464
- if (!found) break;
551
+ matched.push([qi - step.gq - step.n, qi - step.gq]);
552
+ qi = qi - step.gq - step.n;
553
+ si = si - step.gs - step.n;
465
554
  }
466
555
  return { matched, gaps };
467
556
  }
@@ -615,7 +615,23 @@ export async function counterfactualTransfer(
615
615
  fwd !== null && indexOf(answer, fwd, 0) < 0 &&
616
616
  !restatesQuery(query, fwd)
617
617
  ) {
618
- answer = concat2(answer, fwd);
618
+ // THROUGH THE SHARED JOINER, not a bare concatenation.
619
+ //
620
+ // `joinWithBridge` is the composition step every out-of-search assembly
621
+ // shares (multi-topic fusion, CAST's substitution and comparison): it
622
+ // asks the corpus for a learnt connector between the pieces and, on a
623
+ // miss, joins them BARE **and says so** — the `bridgeMiss` step (see
624
+ // resonance.ts). This site bypassed it, and that is the whole of the
625
+ // gluing the study measured: `"Steel is hard"` + `"wet"` came back as
626
+ // `"hardwet"`, `"eva director father"` + `"The father of…"` as
627
+ // `"fatherThe"` — compositions no rationale could show, because the one
628
+ // step that made them left no trace.
629
+ //
630
+ // Routing it through the shared joiner is the instrumentation fix that
631
+ // comes first: a bare join stays possible (the house rule is "joined
632
+ // bare, never silent") but it is now VISIBLE, and an attested connector
633
+ // is used when the corpus has one.
634
+ answer = await joinWithBridge(ctx, answer, fwd);
619
635
  }
620
636
  ctx.trace?.step(
621
637
  "projectCounterfactual",
@@ -98,6 +98,7 @@ export async function resolveConnectors(
98
98
  });
99
99
  const bridgePair = async (l: number, r: number) => {
100
100
  if (l === r || links.has(l + "," + r)) return;
101
+ if (ctx.meter) ctx.meter.coverBridges++;
101
102
  const link = await bridge(ctx, read(ctx, l), read(ctx, r));
102
103
  if (link !== null) links.set(l + "," + r, link);
103
104
  };
@@ -134,6 +135,10 @@ export async function resolveConnectors(
134
135
  // plus one W-quantum of glue per joint — pass that allowance so the
135
136
  // bridge's phrase-scale cap admits the whole learnt run.
136
137
  const allowance = middleBytes + (m + 1) * W;
138
+ if (ctx.meter) {
139
+ ctx.meter.coverBridges++;
140
+ ctx.meter.coverAllowanceBytes += allowance;
141
+ }
137
142
  const interior = await bridge(
138
143
  ctx,
139
144
  first.bytes,
package/src/mind/mind.ts CHANGED
@@ -11,9 +11,12 @@
11
11
 
12
12
  import { cosine, makeKeyring, rng, setVecConfig, Vec } from "../vec.js";
13
13
  import { bindSeat, fold, Sema, Space } from "../sema.js";
14
+ import { sampleCorpus, searchCorpus } from "./corpus.js";
15
+ import type { CorpusPair, CorpusResult } from "./corpus.js";
14
16
  import { Alphabet } from "../alphabet.js";
15
17
  import {
16
18
  bytesToTree,
19
+ contentBoundaries,
17
20
  contentFoldIncremental,
18
21
  Grid,
19
22
  gridToTree,
@@ -146,6 +149,7 @@ interface ConversationData {
146
149
  import type { AttentionRead, MindContext, Recognition } from "./types.js";
147
150
  import { changedNodes, liftAnswer, spliceAll } from "./types.js";
148
151
  import {
152
+ canonResolve as canonResolveImpl,
149
153
  foldTree,
150
154
  gistOf,
151
155
  inputBytes,
@@ -194,10 +198,61 @@ import { type CostReport, Meter } from "../meter.js";
194
198
 
195
199
  // ── MindOptions ───────────────────────────────────────────────────────────
196
200
 
201
+ /** A stored pair as TEXT — the text helper's view of {@link CorpusPair}. */
202
+ export interface CorpusTextPair {
203
+ context: string;
204
+ continuation: string;
205
+ contextId: number;
206
+ continuationId: number;
207
+ matchedBytes: number;
208
+ contextTruncated: boolean;
209
+ continuationTruncated: boolean;
210
+ }
211
+
212
+ /** {@link CorpusResult} as text, plus the prose for why nothing matched. The
213
+ * byte layer reports a STATE; saying it in words belongs to the text layer. */
214
+ export interface CorpusTextResult {
215
+ query: string;
216
+ pairs: CorpusTextPair[];
217
+ resolved: number;
218
+ reached: number;
219
+ totalContexts: number;
220
+ browsed: boolean;
221
+ note?: string;
222
+ }
223
+
224
+ /** What the text helper says when the byte layer reports a miss. */
225
+ const CORPUS_NOTE: Record<string, string> = {
226
+ "nothing-resolved":
227
+ "No trained note sits above the parts of that text the mind recognised. " +
228
+ "It addresses content exactly, so try wording closer to something it was " +
229
+ "actually given — or browse the examples instead.",
230
+ "no-continuations":
231
+ "That text reaches stored nodes, but none of them carries a learnt " +
232
+ "continuation.",
233
+ };
234
+
235
+ /** UTF-8 of bytes for display: reuse {@link decodeText} (the mind's own text
236
+ * conversion), then drop the replacement character a byte-boundary cut leaves
237
+ * behind. Much of a real corpus is non-Latin, so that trailing U+FFFD is the
238
+ * common case, not an exotic one — and it is the ONLY thing added here. */
239
+ function previewCorpusText(bytes: Uint8Array): string {
240
+ return decodeText(bytes).replace(/\uFFFD+$/, "").replace(/\s+/g, " ").trim();
241
+ }
242
+
197
243
  export interface MindOptions {
198
244
  seed?: number;
199
245
  recallQueryK?: number;
200
246
  haloQueryK?: number;
247
+ /** Items one rationale step may itemise — see {@link MindConfig}. */
248
+ rationaleSampleK?: number;
249
+ /** Corpus-reading capacities and budgets — see {@link MindConfig}. */
250
+ corpusLimitMax?: number;
251
+ corpusClimbs?: number;
252
+ corpusContextsPerClimb?: number;
253
+ corpusSampleProbes?: number;
254
+ corpusPreviewBytes?: number;
255
+ corpusSampleFloorBytes?: number;
201
256
  normalizeEpsilon?: number;
202
257
  cosineEpsilon?: number;
203
258
  geometry?: Partial<import("../config.js").GeometryConfig>;
@@ -347,6 +402,20 @@ export class Mind implements MindContext {
347
402
  * `traverse.ts`'s ONE definition (edge or halo, with its response-scoped
348
403
  * cache). The search holds a bare Store and cannot reach that cache itself,
349
404
  * so it asks through this hook; a bare host keeps its raw-store fallback. */
405
+ /** The canonical identity for the search (see GraphSearchHost). */
406
+ canonResolve(bytes: Uint8Array): number | null {
407
+ return canonResolveImpl(this, bytes);
408
+ }
409
+
410
+ /** Feed a search refusal into the rationale (see GraphSearchHost). */
411
+ reportSearch(
412
+ name: string,
413
+ parts: ReadonlyArray<Uint8Array>,
414
+ note: string,
415
+ ): void {
416
+ this.trace?.step(name, parts.map((b) => rItem(b)), [], note);
417
+ }
418
+
350
419
  leadsSomewhere(id: number): boolean {
351
420
  return leadsSomewhere(this, id);
352
421
  }
@@ -374,6 +443,11 @@ export class Mind implements MindContext {
374
443
  * with the most distributional evidence (highest `prevOf` count — the
375
444
  * structural manifestation of its halo). When evidence is equal the
376
445
  * first-inserted edge wins. */
446
+ /** See {@link GraphSearchHost.contentCuts}. */
447
+ contentCuts(bytes: Uint8Array): readonly number[] {
448
+ return contentBoundaries(this.space, bytes);
449
+ }
450
+
377
451
  chooseNext(node: number): number | undefined {
378
452
  return chooseNext(this, node, this._edgeGuide);
379
453
  }
@@ -737,6 +811,55 @@ export class Mind implements MindContext {
737
811
  return decodeText(r.bytes);
738
812
  }
739
813
 
814
+ // ── Reading the trained memory back ─────────────────────────────────────
815
+
816
+ /** Which stored notes does this query REACH? BYTES in, BYTES out — this
817
+ * method has no notion of text or encoding; the text case is
818
+ * {@link searchCorpusText}, which is one caller of this.
819
+ *
820
+ * Exact content addressing through the machinery an answer already uses
821
+ * (see src/mind/corpus.ts): the query's recognised sites are the resolved
822
+ * subtrees, the climb goes up from the biggest, and a result is a context
823
+ * that carries a learnt continuation. Nothing is written and nothing is
824
+ * indexed. */
825
+ searchCorpus(queryBytes: Uint8Array, limit?: number): CorpusResult {
826
+ return searchCorpus(this, queryBytes, limit);
827
+ }
828
+
829
+ /** Browse real pairs. Deterministic: `from` is the caller's own offset in
830
+ * [0,1), so browsing twice with different offsets shows different notes
831
+ * without a random draw. */
832
+ sampleCorpus(limit?: number, from?: number): CorpusResult {
833
+ return sampleCorpus(this, limit, from);
834
+ }
835
+
836
+ /** The TEXT case of {@link searchCorpus}: encode, search, decode. The search
837
+ * itself exists once, in the byte layer above; only the rendering lives
838
+ * here, with the rest of this class's text modality. */
839
+ searchCorpusText(query: string, limit?: number): CorpusTextResult {
840
+ const result = this.searchCorpus(
841
+ new TextEncoder().encode(query),
842
+ limit,
843
+ );
844
+ return {
845
+ query,
846
+ pairs: result.pairs.map((p: CorpusPair): CorpusTextPair => ({
847
+ context: previewCorpusText(p.context),
848
+ continuation: previewCorpusText(p.continuation),
849
+ contextId: p.contextId,
850
+ continuationId: p.continuationId,
851
+ matchedBytes: p.matchedBytes,
852
+ contextTruncated: p.contextTruncated,
853
+ continuationTruncated: p.continuationTruncated,
854
+ })),
855
+ resolved: result.resolved,
856
+ reached: result.reached,
857
+ totalContexts: result.totalContexts,
858
+ browsed: result.browsed,
859
+ note: result.miss === "matched" ? undefined : CORPUS_NOTE[result.miss],
860
+ };
861
+ }
862
+
740
863
  // ── Conversation API ────────────────────────────────────────────────────
741
864
 
742
865
  /** Begin a new conversation, optionally restoring from a previously-saved
@@ -545,12 +545,40 @@ export async function think(
545
545
  ctx.store.nextFirst(id, hubBound(ctx)).map((n) => read(ctx, n))
546
546
  )
547
547
  : [];
548
+ // REPORTABLE, NOT SILENT. A declared-complete grounding ends the derivation
549
+ // here, and that decision is part of the derivation's shape: the reader of a
550
+ // rationale must be able to see that the chain stopped because the mechanism
551
+ // claimed the query WAS the context, not because nothing followed. The step
552
+ // carries the claim, not a re-description of the answer — the extension is
553
+ // skipped, so there is no output item to show.
554
+ if (decided.complete) {
555
+ ctx.trace?.step(
556
+ "completeGrounding",
557
+ [rItem(answer, provenance)],
558
+ [],
559
+ "grounding declared complete — the query IS the context, so " +
560
+ "post-grounding extension is skipped",
561
+ );
562
+ }
563
+ // THE REASONER JUDGES ITS OWN EXTENSIONS BY THE PIPELINE'S REMAINDER, not by
564
+ // the ladder's `accounted` — and by the SAME reading the fuse gate below uses,
565
+ // with the same W floor. `accounted` is a COST quantity (measured: a query
566
+ // fully explained by one computed span plus bridged connectors reports
567
+ // `accounted: []` while nothing is unexplained), and a remainder under one
568
+ // river-fold quantum is bridging punctuation, never a second topic — so it
569
+ // licenses no extension and blocks none.
570
+ const explained: Array<[number, number]> = [
571
+ ...decided.accounted,
572
+ ...pre.computed.map((u): [number, number] => [u.i, u.j]),
573
+ ];
574
+ const uncovered = unexplainedSpans(query.length, explained)
575
+ .filter(([a, b]) => b - a >= ctx.space.maxGroup);
548
576
  const reasoned = decided.complete ? answer : meter
549
577
  ? await meter.time(
550
578
  "reason",
551
- () => reason(ctx, query, answer, preConsumed, pre, voiced),
579
+ () => reason(ctx, query, answer, preConsumed, pre, voiced, uncovered),
552
580
  )
553
- : await reason(ctx, query, answer, preConsumed, pre, voiced);
581
+ : await reason(ctx, query, answer, preConsumed, pre, voiced, uncovered);
554
582
 
555
583
  // Fuse only when the query has a genuine REMAINDER no mechanism's
556
584
  // structural evidence touched at all. `decided.accounted` alone
@@ -568,10 +596,6 @@ export async function think(
568
596
  // observed: a single space between two fully-computed arithmetic spans
569
597
  // ("2+2 3+3") registered as "unaccounted" and pulled in an unrelated
570
598
  // corpus fact, corrupting "4 6" into "4 63".
571
- const explained: Array<[number, number]> = [
572
- ...decided.accounted,
573
- ...pre.computed.map((u): [number, number] => [u.i, u.j]),
574
- ];
575
599
  const remainder = unaccounted(explained);
576
600
  // Whether the winning candidate's entire recognised substance is
577
601
  // COMPUTED — every accounted span exactly a pre.computed span, nothing
@@ -43,6 +43,10 @@ export async function reason(
43
43
  preConsumed: ReadonlySet<number>,
44
44
  pre: Precomputed,
45
45
  voiced: readonly Uint8Array[] = [],
46
+ /** The query material the GROUNDING left uncovered — the cost ladder's own
47
+ * `unaccounted` spans. Only the reasoner's OWN extensions are judged
48
+ * against it; a mechanism carrying its own `used` set owns its shape. */
49
+ uncovered: readonly (readonly [number, number])[] = [],
46
50
  ): Promise<Uint8Array> {
47
51
  // Echo guard: a query that is ITSELF a learnt continuation (some context's
48
52
  // answer) is being asked back at the system — hopping forward from it would
@@ -200,6 +204,57 @@ export async function reason(
200
204
  const fc = await follow(ctx, pivot, qv);
201
205
  consumeAll(pivot);
202
206
  if (fc === null || bytesEqual(fc, cur) || restatesQuery(query, fc)) break;
207
+ // WHOSE EXTENSION IS THIS?
208
+ //
209
+ // `voiced` is what the mechanism WITHHELD (the pipeline sends the used
210
+ // anchors' CONTINUATIONS, not their bytes — see pipeline's own note), so a
211
+ // non-empty `voiced` means exactly what that note says: the grounding came
212
+ // from a mechanism that carries its own short `used` set (cast/join) and
213
+ // therefore owns the shape of its answer. The further terms inside such a
214
+ // seat are legitimately followable — test/29 C3's `Mona Lisa` lives inside
215
+ // the voiced seat and leads on to a fact about neither analog.
216
+ //
217
+ // Every other grounding is ordinary, and an extension of it is the
218
+ // reasoner's own inference: it is taken only while question material the
219
+ // grounding left uncovered remains AND the step carries some of it, judged
220
+ // by the mind's own line between chance and evidence — one W-byte window,
221
+ // no word notion, no character class, no threshold. Measured: the drift's
222
+ // second step (`the Eiffel Tower is in Paris` after `Paris is famous for
223
+ // the Eiffel Tower`) carries no window of `" famous for"` and is refused,
224
+ // while the first carries it. Terminates by a real argument: the uncovered
225
+ // material is finite and each taken extension must carry some of it.
226
+ const producerOwnsShape = voiced.length > 0;
227
+ if (!producerOwnsShape && uncovered.length > 0) {
228
+ const W = ctx.space.maxGroup;
229
+ let progress = false;
230
+ for (const [a, b] of uncovered) {
231
+ for (let i = a; i + W <= b && !progress; i++) {
232
+ if (indexOf(fc, query.subarray(i, i + W), 0) >= 0) progress = true;
233
+ }
234
+ if (progress) break;
235
+ }
236
+ if (!progress) {
237
+ // THE BRAKE, MADE VISIBLE. The reasoner declines a step that carries
238
+ // none of the material the grounding left uncovered — the drift the
239
+ // extension tests pin. A refusal that leaves no trace is the kind of
240
+ // silent cut AGENTS §6 forbids: the rationale is where a reader learns
241
+ // that an extension was declined for want of question material, and
242
+ // where the next person sees why the chain stopped here. Measured with
243
+ // the check disabled, test/110 and test/116 fail — so this brake is the
244
+ // only thing keeping the extension honest until the pivot reports its
245
+ // own accounted spans and the ladder can judge it instead.
246
+ const left = uncovered.reduce((n, [a, b]) => n + (b - a), 0);
247
+ ctx.trace?.step(
248
+ "pivotRefused",
249
+ [rItem(cur, "answer"), rItem(query, "query")],
250
+ uncovered.map(([a, b]) => rItem(query.subarray(a, b), "uncovered")),
251
+ `the step carries none of the question material the grounding left ` +
252
+ `uncovered (${left} byte(s) in ${uncovered.length} span(s)) — refused`,
253
+ );
254
+ break;
255
+ }
256
+ }
257
+ if (ctx.meter) ctx.meter.pivotSteps++;
203
258
  t ??= ctx.trace?.enter("reason", [rItem(startedFrom, "grounded")]);
204
259
  ctx.trace?.step(
205
260
  "pivotStep",
@@ -765,7 +765,15 @@ export function chooseNext(
765
765
  // for the prevCount calls in the loop above, never for extra rItemShort
766
766
  // byte-reads.
767
767
  if (ctx.trace) {
768
- const others = capped.filter((c) => c !== best);
768
+ // A BOUNDED SAMPLE, AND THE COUNT. The step used to carry EVERY candidate
769
+ // it weighed — measured on the trained store, 1559 out-items in one step
770
+ // (hubBound's own size) and 1082 in another (the hub's degree). The
771
+ // rationale's job is to explain the CHOICE, and the count is what says how
772
+ // wide the field was; the declared candidate budget (`recallQueryK`) is what
773
+ // bounds the sample, so no number is invented here.
774
+ const others = capped
775
+ .filter((c) => c !== best)
776
+ .slice(0, ctx.cfg.rationaleSampleK);
769
777
  ctx.trace.step(
770
778
  "disambiguate",
771
779
  [rItemShort(ctx, best, "halo-evidence", bestSupport)],
package/src/mind/types.ts CHANGED
@@ -65,12 +65,38 @@ export interface GraphSearchHost {
65
65
  starts: ReadonlySet<number>;
66
66
  };
67
67
  chooseNext?(node: number): number | undefined;
68
+ /** The boundary positions of `bytes` under the engine's ONE boundary rule
69
+ * (geometry.ts's `contentBoundaries`), or undefined when the host has no
70
+ * space to ask. The join's key is an entity plus a prefix of the tail, and
71
+ * the prefix that names a stored relation ENDS on one of these boundaries —
72
+ * measured, 5 of 5 accepted keys over four join-firing queries, where the
73
+ * byte-by-byte scan spent 153 probes for 14 boundaries. Boundaries are
74
+ * content-defined and STABLE under prefix extension, which is why a corpus
75
+ * key's end is a boundary of the query's own fold of the same bytes. */
76
+ contentCuts?(bytes: Uint8Array): readonly number[];
68
77
  /** The admission predicate — `traverse.ts`'s `leadsSomewhere`, its ONE
69
78
  * definition: does this node bear an edge or a halo? Optional, so a bare
70
79
  * host (a raw Store and nothing else) still works; when present, the search
71
80
  * uses it rather than re-probing the store, which keeps the predicate
72
81
  * single-defined AND memoised on the response-scoped struct cache. */
73
82
  leadsSomewhere?(id: number): boolean;
83
+ /** Report a SEARCH REFUSAL into the rationale — the channel AGENTS §6
84
+ * requires: a callback threaded through a call chain must FEED the
85
+ * rationale, the way `GraphSearch`'s `onDerivation` feeds `traceDerivation`,
86
+ * never a channel of its own. Optional, so a bare host stays silent rather
87
+ * than crashing. */
88
+ reportSearch?(
89
+ name: string,
90
+ parts: ReadonlyArray<Uint8Array>,
91
+ note: string,
92
+ ): void;
93
+ /** The CANONICAL resolver ({@link canonResolve}), optional like
94
+ * {@link leadsSomewhere}. The store's keys were written through the
95
+ * canonical fold, so a fact's `Gustaf Molander` and the deposited
96
+ * `gustaf molander` are the SAME node (measured inside a response: the
97
+ * canonical resolver maps the surface form to the deposited node while a raw
98
+ * resolve returns null). A bare host falls back to the plain probe. */
99
+ canonResolve?(bytes: Uint8Array): number | null;
74
100
  }
75
101
 
76
102
  // ═══════════════════════════════════════════════════════════════════════════