@hviana/sema 0.8.1 → 0.8.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (114) hide show
  1. package/AGENTS.md +29 -29
  2. package/TRADEMARKS.md +0 -1
  3. package/dist/src/config.d.ts +28 -0
  4. package/dist/src/config.js +20 -0
  5. package/dist/src/geometry.d.ts +21 -0
  6. package/dist/src/geometry.js +21 -0
  7. package/dist/src/meter.d.ts +76 -0
  8. package/dist/src/meter.js +95 -0
  9. package/dist/src/mind/attention.d.ts +4 -0
  10. package/dist/src/mind/attention.js +165 -16
  11. package/dist/src/mind/canonical.d.ts +16 -0
  12. package/dist/src/mind/canonical.js +41 -0
  13. package/dist/src/mind/corpus.d.ts +40 -0
  14. package/dist/src/mind/corpus.js +149 -0
  15. package/dist/src/mind/graph-search.d.ts +7 -0
  16. package/dist/src/mind/graph-search.js +254 -24
  17. package/dist/src/mind/index.d.ts +3 -1
  18. package/dist/src/mind/index.js +1 -0
  19. package/dist/src/mind/match.d.ts +9 -4
  20. package/dist/src/mind/match.js +147 -61
  21. package/dist/src/mind/mechanisms/cast.js +19 -3
  22. package/dist/src/mind/mechanisms/confluence.js +24 -0
  23. package/dist/src/mind/mechanisms/cover.js +6 -0
  24. package/dist/src/mind/mechanisms/recall.js +32 -4
  25. package/dist/src/mind/mind.d.ts +57 -0
  26. package/dist/src/mind/mind.js +72 -1
  27. package/dist/src/mind/pipeline-mechanism.d.ts +7 -0
  28. package/dist/src/mind/pipeline.js +66 -20
  29. package/dist/src/mind/primitives.js +9 -1
  30. package/dist/src/mind/rationale.d.ts +28 -1
  31. package/dist/src/mind/rationale.js +22 -1
  32. package/dist/src/mind/reasoning.d.ts +25 -3
  33. package/dist/src/mind/reasoning.js +125 -20
  34. package/dist/src/mind/recognition.js +4 -8
  35. package/dist/src/mind/resonance.js +20 -1
  36. package/dist/src/mind/trace.js +1 -0
  37. package/dist/src/mind/traverse.js +15 -3
  38. package/dist/src/mind/types.d.ts +49 -4
  39. package/docs/INVARIANTS.md +2 -2
  40. package/docs/architecture/bounded-reads.md +1 -1
  41. package/docs/architecture/commonality.md +2 -2
  42. package/docs/architecture/cost-model.md +2 -2
  43. package/docs/architecture/determinism.md +7 -7
  44. package/docs/architecture/match-project.md +2 -3
  45. package/docs/architecture/mechanism-market.md +10 -10
  46. package/docs/architecture/meter.md +5 -5
  47. package/docs/architecture/store.md +3 -3
  48. package/docs/failures/tempting-but-wrong.md +34 -6
  49. package/docs/harness/gates.md +2 -2
  50. package/docs/mechanisms/cast.md +2 -2
  51. package/docs/mechanisms/cover.md +2 -3
  52. package/docs/mechanisms/extraction.md +7 -7
  53. package/docs/mechanisms/recall.md +8 -9
  54. package/jsr.json +1 -1
  55. package/package.json +1 -1
  56. package/src/alu/README.md +11 -12
  57. package/src/config.ts +48 -0
  58. package/src/geometry.ts +21 -0
  59. package/src/meter.ts +98 -0
  60. package/src/mind/attention.ts +167 -16
  61. package/src/mind/canonical.ts +43 -0
  62. package/src/mind/corpus.ts +202 -0
  63. package/src/mind/graph-search.ts +277 -23
  64. package/src/mind/index.ts +8 -1
  65. package/src/mind/match.ts +148 -57
  66. package/src/mind/mechanisms/cast.ts +20 -2
  67. package/src/mind/mechanisms/confluence.ts +24 -0
  68. package/src/mind/mechanisms/cover.ts +5 -0
  69. package/src/mind/mechanisms/recall.ts +32 -4
  70. package/src/mind/mind.ts +125 -0
  71. package/src/mind/pipeline-mechanism.ts +7 -0
  72. package/src/mind/pipeline.ts +79 -22
  73. package/src/mind/primitives.ts +9 -1
  74. package/src/mind/rationale.ts +35 -1
  75. package/src/mind/reasoning.ts +145 -13
  76. package/src/mind/recognition.ts +4 -8
  77. package/src/mind/resonance.ts +19 -1
  78. package/src/mind/trace.ts +1 -0
  79. package/src/mind/traverse.ts +16 -6
  80. package/src/mind/types.ts +53 -4
  81. package/test/100-complete-grounding-trace.test.mjs +109 -0
  82. package/test/101-alignment-gap-bound.test.mjs +106 -0
  83. package/test/102-production-composes-at-scale.test.mjs +110 -0
  84. package/test/103-alignment-gap-budget.test.mjs +89 -0
  85. package/test/104-composition-is-reported.test.mjs +90 -0
  86. package/test/105-derive-through-reports-its-refusal.test.mjs +137 -0
  87. package/test/106-the-join-fires.test.mjs +94 -0
  88. package/test/107-the-join-is-counted.test.mjs +81 -0
  89. package/test/108-the-join-chains.test.mjs +78 -0
  90. package/test/109-the-pivot-is-counted.test.mjs +60 -0
  91. package/test/110-the-reasoner-stops-when-the-question-is-answered.test.mjs +91 -0
  92. package/test/111-the-cover-assembly-is-counted.test.mjs +74 -0
  93. package/test/112-the-exploration-does-not-grow-with-the-hub.test.mjs +89 -0
  94. package/test/113-the-rationale-payload-is-bounded.test.mjs +84 -0
  95. package/test/114-alignment-budget-is-per-sweep.test.mjs +93 -0
  96. package/test/116-the-extension-is-gated-by-the-pipelines-own-remainder.test.mjs +100 -0
  97. package/test/117-corpus-search.test.mjs +171 -0
  98. package/test/118-the-join-reaches-a-key-off-the-cut.test.mjs +74 -0
  99. package/test/119-the-work-does-not-grow-with-the-corpus.test.mjs +122 -0
  100. package/test/120-composition-is-consequence.test.mjs +132 -0
  101. package/test/121-the-extension-does-not-grow-with-the-corpus.test.mjs +128 -0
  102. package/test/122-the-climb-search-does-not-grow-with-the-corpus.test.mjs +117 -0
  103. package/test/123-the-paired-formulas-agree.test.mjs +90 -0
  104. package/test/125-the-post-grounding-branch-publishes-its-operand.test.mjs +51 -0
  105. package/test/126-the-pipeline-does-not-name-mechanisms.test.mjs +42 -0
  106. package/test/128-the-leads-somewhere-pair-agrees.test.mjs +83 -0
  107. package/test/129-the-trace-payload-shape.test.mjs +164 -0
  108. package/test/14-scaling.test.mjs +10 -7
  109. package/test/32-confluence.test.mjs +68 -0
  110. package/test/38-reason-restate-guard.test.mjs +8 -2
  111. package/test/43-cast-analog-seat.test.mjs +10 -0
  112. package/test/55-cost-meter.test.mjs +859 -0
  113. package/test/76-reference-binding.test.mjs +6 -1
  114. package/test/89-completion-recursion.test.mjs +30 -5
package/src/mind/match.ts CHANGED
@@ -38,7 +38,7 @@ import {
38
38
  identityBar,
39
39
  significanceBar,
40
40
  } from "../geometry.js";
41
- import { bytesEqual, indexOf } from "../bytes.js";
41
+ import { bytesEqual, indexOf, latin1 } from "../bytes.js";
42
42
  import type { MindContext } from "./types.js";
43
43
  import { chainReach, leafIdRun } from "./canonical.js";
44
44
  import { foldTree, gistOf, perceive, read, resolve } from "./primitives.js";
@@ -54,6 +54,7 @@ import {
54
54
  sharedReachMemo,
55
55
  } from "./traverse.js";
56
56
  import { recognise, segment } from "./recognition.js";
57
+ import { rItem } from "./trace.js";
57
58
  import type { Site } from "./graph-search.js";
58
59
 
59
60
  // ═══════════════════════════════════════════════════════════════════════════
@@ -360,9 +361,14 @@ export interface AlignGap {
360
361
 
361
362
  /** Extend a seed match (query offset qo ↔ candidate offset co) to its maximal
362
363
  * common run, then walk outward in both directions collecting further common
363
- * runs of at least W bytes across bounded mismatch gaps (each side ≤
364
- * chainReach). Returns the matched query spans and the mismatch pairs
365
- * between consecutive runs.
364
+ * runs of at least W bytes across mismatch gaps. Each gap's LENGTH is the
365
+ * pair's own extent (a gap cannot be longer than the bytes it spans) and the
366
+ * sweep's WORK is proportional to the bytes a run spans (the context's windows
367
+ * are indexed once, then the query's are walked) — the arity bound
368
+ * (`chainReach`) used to cap BOTH, and truncated every learned frame whose
369
+ * slot was longer. Each sweep owns its own budget, so an exhausted right
370
+ * sweep never starves the left one. Returns the matched query spans and the
371
+ * mismatch pairs between consecutive runs.
366
372
  *
367
373
  * This is the SEEDED aligner, distinct from {@link alignRuns}: that one finds
368
374
  * every run two structures share anywhere (a weave), this one reads two
@@ -382,7 +388,21 @@ export function alignAround(
382
388
  co: number,
383
389
  ): { matched: Array<[number, number]>; gaps: AlignGap[] } {
384
390
  const W = ctx.space.maxGroup;
385
- const reachCap = chainReach(W);
391
+ // THE GAP LENGTH IS THE PAIR'S OWN EXTENT; THE WORK IS BUDGETED.
392
+ //
393
+ // The sweep walks (queryGap, contextGap) pairs by ASCENDING total, so reaching
394
+ // a gap of size G costs about G²/2 pairs. Bounding the LENGTH by the write
395
+ // side's arity (`chainReach(W)` = 16) therefore truncated every learned frame
396
+ // whose slot is longer — measured: `bindReference` reported the cap at 18, 24,
397
+ // 30 and 36 bytes and `recall` answered with ANOTHER instance's filler — while
398
+ // removing the bound outright took the corpus-cost guard (test/89) from
399
+ // milliseconds to 68 seconds.
400
+ //
401
+ // Bounding the PAIRS keeps a call's cost constant however long the pair is,
402
+ // and the ascending order means an exhausted budget drops the FAR
403
+ // continuations and never the near ones — the same degradation recognition.ts
404
+ // documents for its canon budget. Length and work are different questions;
405
+ // this is the one place they were conflated.
386
406
  // Maximal run around the seed.
387
407
  let qs = qo, ss = co;
388
408
  while (qs > 0 && ss > 0 && q[qs - 1] === c[ss - 1]) {
@@ -396,8 +416,60 @@ export function alignAround(
396
416
  }
397
417
  const matched: Array<[number, number]> = [[qs, qe]];
398
418
  const gaps: AlignGap[] = [];
399
- // The next common run of ≥ W bytes past (qi, si), with each side's gap
400
- // bounded by chainReach; smallest total gap wins (nearest continuation).
419
+ // THE SWEEP IS STRUCTURAL, NOT ENUMERATIVE.
420
+ //
421
+ // The criterion is unchanged: the next common run, MINIMUM TOTAL GAP, ties to
422
+ // the smaller query gap. What changed is how it is found. Enumerating
423
+ // (queryGap, contextGap) pairs by ascending total reaches a run at total t in
424
+ // about t²/2 pairs — and that quadratic shape, not the reach, was the cost
425
+ // problem: capping the pairs dropped reach (a legitimate 24-byte slot stopped
426
+ // being found), while leaving them uncapped cost 68 seconds on the corpus
427
+ // guard. Neither is the answer, because the answer is the algorithm.
428
+ //
429
+ // The context's windows are indexed ONCE, for lengths 1..W — W being the
430
+ // geometry's own unit of composition, so nothing is chosen here. Each step
431
+ // then walks the query's windows outward from the anchor: for a given query
432
+ // gap the nearest context gap that continues a run is one O(1) lookup, and the
433
+ // walk stops the moment the query gap alone exceeds the best total already
434
+ // found. So the work is proportional to the bytes the run SPANS. No budget,
435
+ // no cap, no number: a long slot is reached, and its price is already the
436
+ // ladder's (its bytes are unaccounted, so the search pays PASS per byte).
437
+ const index: Array<Map<string, number[]>> = [];
438
+ for (let len = 1; len <= W; len++) {
439
+ const m = new Map<string, number[]>();
440
+ for (let o = 0; o + len <= c.length; o++) {
441
+ const key = latin1(c.subarray(o, o + len));
442
+ const at = m.get(key);
443
+ if (at === undefined) m.set(key, [o]);
444
+ else at.push(o);
445
+ }
446
+ index.push(m);
447
+ }
448
+ /** Smallest listed offset at or after `from`, or -1. */
449
+ const fromAt = (list: number[], from: number): number => {
450
+ let lo = 0, hi = list.length - 1, best = -1;
451
+ while (lo <= hi) {
452
+ const mid = (lo + hi) >> 1;
453
+ if (list[mid] >= from) {
454
+ best = list[mid];
455
+ hi = mid - 1;
456
+ } else lo = mid + 1;
457
+ }
458
+ return best;
459
+ };
460
+ /** Largest listed offset at or before `to`, or -1. */
461
+ const toAt = (list: number[], to: number): number => {
462
+ let lo = 0, hi = list.length - 1, best = -1;
463
+ while (lo <= hi) {
464
+ const mid = (lo + hi) >> 1;
465
+ if (list[mid] <= to) {
466
+ best = list[mid];
467
+ lo = mid + 1;
468
+ } else hi = mid - 1;
469
+ }
470
+ return best;
471
+ };
472
+ /** Length of the common run STARTING at (qi, si). */
401
473
  const runLenAt = (qi: number, si: number): number => {
402
474
  let n = 0;
403
475
  while (qi + n < q.length && si + n < c.length && q[qi + n] === c[si + n]) {
@@ -405,63 +477,80 @@ export function alignAround(
405
477
  }
406
478
  return n;
407
479
  };
408
- // RIGHT sweep.
409
- let qi = qe, si = se;
410
- for (;;) {
411
- let found = false;
412
- for (let total = 1; total <= 2 * reachCap && !found; total++) {
413
- for (let gq = 0; gq <= Math.min(total, reachCap); gq++) {
414
- const gs = total - gq;
415
- if (gs > reachCap) continue;
416
- if (qi + gq >= q.length || si + gs >= c.length) continue;
417
- const n = runLenAt(qi + gq, si + gs);
418
- if (n >= W || qi + gq + n === q.length) {
419
- if (n === 0) continue;
420
- if (gq > 0 || gs > 0) {
421
- gaps.push({ qs: qi, qe: qi + gq, cs: si, ce: si + gs });
480
+ /** Length of the common run ENDING at (qi, si). */
481
+ const runLenBefore = (qi: number, si: number): number => {
482
+ let n = 0;
483
+ while (n < qi && n < si && q[qi - 1 - n] === c[si - 1 - n]) n++;
484
+ return n;
485
+ };
486
+ /** The next run outward from an anchor, or null when the bytes run out. */
487
+ const nextRun = (
488
+ qi: number,
489
+ si: number,
490
+ forward: boolean,
491
+ ): { gq: number; gs: number; n: number } | null => {
492
+ const qLim = forward ? q.length - qi : qi;
493
+ let best: { gq: number; gs: number; n: number } | null = null;
494
+ for (let gq = 0; gq < qLim; gq++) {
495
+ // No later query gap can beat a total already found.
496
+ if (best !== null && gq > best.gq + best.gs) break;
497
+ const left = qLim - gq;
498
+ // A run of >= W bytes, or — when the query itself ends inside one window —
499
+ // the run that REACHES that end. Exactly the acceptance the sweep had.
500
+ const lens = left >= W ? [W] : [left];
501
+ for (const len of lens) {
502
+ const key = latin1(
503
+ q.subarray(
504
+ forward ? qi + gq : qi - gq - len,
505
+ forward ? qi + gq + len : qi - gq,
506
+ ),
507
+ );
508
+ const list = index[len - 1].get(key);
509
+ if (list === undefined) continue;
510
+ const o = forward ? fromAt(list, si) : toAt(list, si - len);
511
+ if (o < 0) continue;
512
+ const n = forward
513
+ ? runLenAt(qi + gq, o)
514
+ : runLenBefore(qi - gq, o + len);
515
+ if (n < 1) continue;
516
+ if (
517
+ forward ? n >= W || qi + gq + n === q.length : n >= W || n === qi - gq
518
+ ) {
519
+ const gs = forward ? o - si : si - len - o;
520
+ if (best === null || gq + gs < best.gq + best.gs) {
521
+ best = { gq, gs, n };
422
522
  }
423
- matched.push([qi + gq, qi + gq + n]);
424
- qi = qi + gq + n;
425
- si = si + gs + n;
426
- found = true;
427
523
  break;
428
524
  }
429
525
  }
430
526
  }
431
- if (!found) break;
527
+ return best;
528
+ };
529
+ // RIGHT sweep.
530
+ let qi = qe, si = se;
531
+ for (;;) {
532
+ const step = nextRun(qi, si, true);
533
+ if (step === null) break;
534
+ if (step.gq > 0 || step.gs > 0) {
535
+ gaps.push({ qs: qi, qe: qi + step.gq, cs: si, ce: si + step.gs });
536
+ }
537
+ matched.push([qi + step.gq, qi + step.gq + step.n]);
538
+ qi = qi + step.gq + step.n;
539
+ si = si + step.gs + step.n;
432
540
  }
433
- // LEFT sweep (mirror).
541
+ // LEFT sweep (mirror): an independent walk, so an exhausted right side can
542
+ // never starve it (pinned by test/114).
434
543
  qi = qs;
435
544
  si = ss;
436
545
  for (;;) {
437
- let found = false;
438
- for (let total = 1; total <= 2 * reachCap && !found; total++) {
439
- for (let gq = 0; gq <= Math.min(total, reachCap); gq++) {
440
- const gs = total - gq;
441
- if (gs > reachCap) continue;
442
- if (qi - gq <= 0 || si - gs <= 0) continue;
443
- // Run ENDING at (qi - gq, si - gs).
444
- let n = 0;
445
- while (
446
- n < qi - gq && n < si - gs &&
447
- q[qi - gq - 1 - n] === c[si - gs - 1 - n]
448
- ) {
449
- n++;
450
- }
451
- if (n >= W || n === qi - gq) {
452
- if (n === 0) continue;
453
- if (gq > 0 || gs > 0) {
454
- gaps.push({ qs: qi - gq, qe: qi, cs: si - gs, ce: si });
455
- }
456
- matched.push([qi - gq - n, qi - gq]);
457
- qi = qi - gq - n;
458
- si = si - gs - n;
459
- found = true;
460
- break;
461
- }
462
- }
546
+ const step = nextRun(qi, si, false);
547
+ if (step === null) break;
548
+ if (step.gq > 0 || step.gs > 0) {
549
+ gaps.push({ qs: qi - step.gq, qe: qi, cs: si - step.gs, ce: si });
463
550
  }
464
- if (!found) break;
551
+ matched.push([qi - step.gq - step.n, qi - step.gq]);
552
+ qi = qi - step.gq - step.n;
553
+ si = si - step.gs - step.n;
465
554
  }
466
555
  return { matched, gaps };
467
556
  }
@@ -634,7 +723,7 @@ export function frameSlots(
634
723
  * ANCHOR that the query displaced. Neither implies the other, and the
635
724
  * observed failures pass the restatement guard cleanly.
636
725
  *
637
- * Three conditions, all byte-exact and all necessary:
726
+ * Four conditions, all byte-exact and all necessary:
638
727
  *
639
728
  * 1. the query and the anchor must be ONE STRUCTURE — what they share has to
640
729
  * dominate the query, or the query is not a variant of the anchor at all
@@ -722,8 +811,10 @@ export function substituteAll(
722
811
  const usable = pairs.filter((p) => p.needle.length > 0);
723
812
  if (usable.length === 0) return hay;
724
813
  // Longest needle first, so a needle that is a prefix of another can never
725
- // pre-empt it. Ties cannot arise: an instance whose fillers are not
726
- // pairwise distinct is refused by frameSlots.
814
+ // pre-empt it. Ties cannot arise: a consumer that VOICES checks the
815
+ // fillers pairwise with `distinct` and refuses such an instance itself —
816
+ // `frameSlots` reports and does not judge (see its own doc), so the refusal
817
+ // lives with the mechanism that needs it, not here.
727
818
  const order = [...usable].sort((a, b) => b.needle.length - a.needle.length);
728
819
  const out: number[] = [];
729
820
  let i = 0;
@@ -615,7 +615,23 @@ export async function counterfactualTransfer(
615
615
  fwd !== null && indexOf(answer, fwd, 0) < 0 &&
616
616
  !restatesQuery(query, fwd)
617
617
  ) {
618
- answer = concat2(answer, fwd);
618
+ // THROUGH THE SHARED JOINER, not a bare concatenation.
619
+ //
620
+ // `joinWithBridge` is the composition step every out-of-search assembly
621
+ // shares (multi-topic fusion, CAST's substitution and comparison): it
622
+ // asks the corpus for a learnt connector between the pieces and, on a
623
+ // miss, joins them BARE **and says so** — the `bridgeMiss` step (see
624
+ // resonance.ts). This site bypassed it, and that is the whole of the
625
+ // gluing the study measured: `"Steel is hard"` + `"wet"` came back as
626
+ // `"hardwet"`, `"eva director father"` + `"The father of…"` as
627
+ // `"fatherThe"` — compositions no rationale could show, because the one
628
+ // step that made them left no trace.
629
+ //
630
+ // Routing it through the shared joiner is the instrumentation fix that
631
+ // comes first: a bare join stays possible (the house rule is "joined
632
+ // bare, never silent") but it is now VISIBLE, and an attested connector
633
+ // is used when the corpus has one.
634
+ answer = await joinWithBridge(ctx, answer, fwd);
619
635
  }
620
636
  ctx.trace?.step(
621
637
  "projectCounterfactual",
@@ -851,7 +867,9 @@ export async function counterfactualTransfer(
851
867
  // grounds") — fine for ORIENTING mechanisms, not for voicing learnt
852
868
  // content the query never asked about. Computed once here; both the
853
869
  // hub fallback below and the comparison gate consume it.
854
- const rootTrusted = roots.some((r) => r.vote >= consensusFloor(corpusN(ctx)));
870
+ const rootTrusted = roots.some((r) =>
871
+ r.idfVote >= consensusFloor(corpusN(ctx))
872
+ ); // the IDF sum: the bar's own quantity
855
873
  // The context that ESTABLISHES a filler — the same reverse context, under
856
874
  // the same naming test, `seatOfNode` uses to VOICE an analog (a predecessor
857
875
  // whose bytes CONTAIN the node's: it names or describes it, rather than
@@ -146,6 +146,30 @@ export async function confluenceJoin(
146
146
  const bindsAConstituent = (cover: Array<[number, number]>): boolean =>
147
147
  cover.some(([cs, ce]) => ce - cs >= 2 * W);
148
148
 
149
+ // THE VOTE ENTERS AS ORDER, NEVER AS A BAR. This is the only one of the
150
+ // climb's four consumers (recall, fuseAttention, cast, here) that uses the
151
+ // evidence's MAGNITUDE without a floor, and it is legitimate by construction:
152
+ // `ranked` answers "which anchor is stronger" — a question about votes, so the
153
+ // comparison stays within one dimension — and the vote is otherwise only
154
+ // REPORTED (Stream.vote travels to the rationale's constraint nodes). What
155
+ // actually SELECTS a constraint is byte-structural and never the magnitude: a
156
+ // run of at least 2W (`bindsAConstituent`, with its accidental-sharing
157
+ // counter-examples above), disjoint covers (`disjoint`), and scaffolding never
158
+ // binds at all (`dominates(reachOf(…), N)`). The MEET such a stream may
159
+ // produce is selected the same way: a span shorter than 2W is rejected, and
160
+ // the winner is the one with the smallest `reach` (ties broken by the longer
161
+ // span) — a corpus quantity and bytes, never the vote, which appears only in
162
+ // the trace item.
163
+ // binds at all (`dominates(reachOf(…), N)`). The only cut in this loop is a
164
+ // BUDGET, and it is measured: stopping the scan at 2W anchors saves 50-70% of
165
+ // confluence's cost on non-conjunctive queries while preserving every genuinely
166
+ // conjunctive case, whose top anchors ARE its constraints.
167
+ // MEASURED (this goal, on THIS file's own conjunctive fixture): the two
168
+ // streams appear at ranks 1 and 4 against a budget of 2W = 8, on a query whose
169
+ // `ranked` is 9 — so the cut IS live (it would have returned null at the 8th
170
+ // anchor) and it does NOT prune the case it exists to protect. The other
171
+ // conjunctive fixture (the Leonardo one) finds them at ranks 0 and 1. Scope:
172
+ // these are the repo's conjunctive fixtures, and no more.
149
173
  const streams: Stream[] = [];
150
174
  const rankedCapped = ranked.length > pre.k ? ranked.slice(0, pre.k) : ranked;
151
175
  // CONJUNCTIVITY EARLY-EXIT: a conjunctive query's top-ranked anchors
@@ -98,6 +98,7 @@ export async function resolveConnectors(
98
98
  });
99
99
  const bridgePair = async (l: number, r: number) => {
100
100
  if (l === r || links.has(l + "," + r)) return;
101
+ if (ctx.meter) ctx.meter.coverBridges++;
101
102
  const link = await bridge(ctx, read(ctx, l), read(ctx, r));
102
103
  if (link !== null) links.set(l + "," + r, link);
103
104
  };
@@ -134,6 +135,10 @@ export async function resolveConnectors(
134
135
  // plus one W-quantum of glue per joint — pass that allowance so the
135
136
  // bridge's phrase-scale cap admits the whole learnt run.
136
137
  const allowance = middleBytes + (m + 1) * W;
138
+ if (ctx.meter) {
139
+ ctx.meter.coverBridges++;
140
+ ctx.meter.coverAllowanceBytes += allowance;
141
+ }
137
142
  const interior = await bridge(
138
143
  ctx,
139
144
  first.bytes,
@@ -276,9 +276,36 @@ export async function recallByResonance(
276
276
  // consensus", while breadth is the SCALE-INVARIANT reading — "a point whose
277
277
  // breadth clears `dominates` (> half the query's regions corroborate it) is
278
278
  // real consensus; one that does not is a coincidental single-region echo".
279
- // Attention.peak's contract makes the same point from the other side:
280
- // comparing a POOLED SUM against a floor that prices ONE region's evidence
281
- // is a dimensional error.
279
+ // THIS USED TO CLAIM A DIMENSIONAL ERROR, AND THAT CLAIM WAS FALSE.
280
+ // It read: "comparing a POOLED SUM against a floor that prices ONE region's
281
+ // evidence is a dimensional error." `consensusFloor` is not priced for one
282
+ // region: thresholds.md §2 derives it as the POOLED-vote significance floor —
283
+ // "each region contributes at most ln(N/c) <= ln(N); ln(N)+1/2 demands ..." —
284
+ // and attention.ts says the same where it builds the vote ("the scale
285
+ // consensusFloor is derived for"). The comparison is in ONE dimension, and
286
+ // it is so because the climb WEIGHTS BY IDF: `wf` in voteRegions is
287
+ // `direct ? df : combined ? idf + df : idf`, and the engine only ever runs the
288
+ // last one (DFMode's default "inverse", the mode every non-test caller uses —
289
+ // `direct` and `combined` are exercised by test/24 and test/27 only, and
290
+ // test/24 pins that their votes DO differ). In those two the sum would leave
291
+ // the floor's dimension and the floor would need re-deriving.
292
+ //
293
+ // What the OR below is really for is SCALE, not dimension (the paragraph
294
+ // above says it): a vote that clears ln(N)+1/2 means "strong" on a small store
295
+ // and "weak" on a large one for the same genuine consensus, so the
296
+ // scale-invariant breadth reading is added beside it.
297
+ //
298
+ // AND THE PREMISE IS IDF. The deviation in the other two weighting modes is
299
+ // TWO-SIDED and DERIVED: `direct` DEFLATES a region (ln(1+c) < ln(N/c) for
300
+ // small c) and `combined` INFLATES it (ln N + ln(1+1/c)), both by at most
301
+ // `ln 2` — see `geometry.ts`'s `consensusFloor`, where the bound lives.
302
+ // MEASURED on 8 anchors across 5 queries, running the same climb in all
303
+ // three modes: ZERO gate inversions — every anchor's `vote >= floor` verdict
304
+ // is the same in `inverse`, `direct` and `combined`, even where the readings
305
+ // straddle the floor on opposite sides (#148: inverse 3.39, combined 4.71
306
+ // above it, direct 1.31 below). Pinned by test/55's test 20. The bar is not
307
+ // re-derived for those modes because nothing reachable needs it; the premise
308
+ // is IDF, and that is now written where the gate reads it.
282
309
  //
283
310
  // Measured on the 15.7M-node store (N=325,615, so the old floor was 13.19).
284
311
  // The absolute vote cannot separate right from wrong at this scale, and the
@@ -348,7 +375,7 @@ export async function recallByResonance(
348
375
  if (
349
376
  forest.length > 0 &&
350
377
  !allWindowsAreScaffolding(ctx, query) &&
351
- (forest[0].vote >= minVote ||
378
+ (forest[0].idfVote >= minVote || // the IDF sum: the bar's own quantity
352
379
  (dominates(forest[0].breadth, 1) && forest[0].peak > Math.LN2))
353
380
  ) {
354
381
  const g = await project(ctx, forest[0].anchor, queryGist);
@@ -602,6 +629,7 @@ export const recallMechanism: PipelineMechanism = {
602
629
  moves: r.moves,
603
630
  unexplained: r.unexplained,
604
631
  provenance: r.echoed ? "recall-echo" : "recall",
632
+ used: new Set<number>(),
605
633
  ...(r.complete ? { complete: true } : {}),
606
634
  }];
607
635
  },
package/src/mind/mind.ts CHANGED
@@ -11,6 +11,8 @@
11
11
 
12
12
  import { cosine, makeKeyring, rng, setVecConfig, Vec } from "../vec.js";
13
13
  import { bindSeat, fold, Sema, Space } from "../sema.js";
14
+ import { sampleCorpus, searchCorpus } from "./corpus.js";
15
+ import type { CorpusPair, CorpusResult } from "./corpus.js";
14
16
  import { Alphabet } from "../alphabet.js";
15
17
  import {
16
18
  bytesToTree,
@@ -21,6 +23,7 @@ import {
21
23
  reachThreshold,
22
24
  stackGrids,
23
25
  } from "../geometry.js";
26
+ import { keyEnds } from "./canonical.js";
24
27
  import type { ContentFold } from "../geometry.js";
25
28
  import { BoundedMap, type Store } from "../store.js";
26
29
  import { SQliteStore } from "../store-sqlite.js";
@@ -146,6 +149,7 @@ interface ConversationData {
146
149
  import type { AttentionRead, MindContext, Recognition } from "./types.js";
147
150
  import { changedNodes, liftAnswer, spliceAll } from "./types.js";
148
151
  import {
152
+ canonResolve as canonResolveImpl,
149
153
  foldTree,
150
154
  gistOf,
151
155
  inputBytes,
@@ -194,10 +198,63 @@ import { type CostReport, Meter } from "../meter.js";
194
198
 
195
199
  // ── MindOptions ───────────────────────────────────────────────────────────
196
200
 
201
+ /** A stored pair as TEXT — the text helper's view of {@link CorpusPair}. */
202
+ export interface CorpusTextPair {
203
+ context: string;
204
+ continuation: string;
205
+ contextId: number;
206
+ continuationId: number;
207
+ matchedBytes: number;
208
+ contextTruncated: boolean;
209
+ continuationTruncated: boolean;
210
+ }
211
+
212
+ /** {@link CorpusResult} as text, plus the prose for why nothing matched. The
213
+ * byte layer reports a STATE; saying it in words belongs to the text layer. */
214
+ export interface CorpusTextResult {
215
+ query: string;
216
+ pairs: CorpusTextPair[];
217
+ resolved: number;
218
+ reached: number;
219
+ totalContexts: number;
220
+ browsed: boolean;
221
+ note?: string;
222
+ }
223
+
224
+ /** What the text helper says when the byte layer reports a miss. */
225
+ const CORPUS_NOTE: Record<string, string> = {
226
+ "nothing-resolved":
227
+ "No trained note sits above the parts of that text the mind recognised. " +
228
+ "It addresses content exactly, so try wording closer to something it was " +
229
+ "actually given — or browse the examples instead.",
230
+ "no-continuations":
231
+ "That text reaches stored nodes, but none of them carries a learnt " +
232
+ "continuation.",
233
+ };
234
+
235
+ /** UTF-8 of bytes for display: reuse {@link decodeText} (the mind's own text
236
+ * conversion), then drop the replacement character a byte-boundary cut leaves
237
+ * behind. Much of a real corpus is non-Latin, so that trailing U+FFFD is the
238
+ * common case, not an exotic one — and it is the ONLY thing added here. */
239
+ function previewCorpusText(bytes: Uint8Array): string {
240
+ return decodeText(bytes).replace(/\uFFFD+$/, "").replace(/\s+/g, " ").trim();
241
+ }
242
+
197
243
  export interface MindOptions {
198
244
  seed?: number;
199
245
  recallQueryK?: number;
200
246
  haloQueryK?: number;
247
+ /** Branch nodes the pivot sweep may probe — see {@link MindConfig}. */
248
+ pivotProbeK?: number;
249
+ /** Items one rationale step may itemise — see {@link MindConfig}. */
250
+ rationaleSampleK?: number;
251
+ /** Corpus-reading capacities and budgets — see {@link MindConfig}. */
252
+ corpusLimitMax?: number;
253
+ corpusClimbs?: number;
254
+ corpusContextsPerClimb?: number;
255
+ corpusSampleProbes?: number;
256
+ corpusPreviewBytes?: number;
257
+ corpusSampleFloorBytes?: number;
201
258
  normalizeEpsilon?: number;
202
259
  cosineEpsilon?: number;
203
260
  geometry?: Partial<import("../config.js").GeometryConfig>;
@@ -347,6 +404,20 @@ export class Mind implements MindContext {
347
404
  * `traverse.ts`'s ONE definition (edge or halo, with its response-scoped
348
405
  * cache). The search holds a bare Store and cannot reach that cache itself,
349
406
  * so it asks through this hook; a bare host keeps its raw-store fallback. */
407
+ /** The canonical identity for the search (see GraphSearchHost). */
408
+ canonResolve(bytes: Uint8Array): number | null {
409
+ return canonResolveImpl(this, bytes);
410
+ }
411
+
412
+ /** Feed a search refusal into the rationale (see GraphSearchHost). */
413
+ reportSearch(
414
+ name: string,
415
+ parts: ReadonlyArray<Uint8Array>,
416
+ note: string,
417
+ ): void {
418
+ this.trace?.step(name, parts.map((b) => rItem(b)), [], note);
419
+ }
420
+
350
421
  leadsSomewhere(id: number): boolean {
351
422
  return leadsSomewhere(this, id);
352
423
  }
@@ -374,6 +445,11 @@ export class Mind implements MindContext {
374
445
  * with the most distributional evidence (highest `prevOf` count — the
375
446
  * structural manifestation of its halo). When evidence is equal the
376
447
  * first-inserted edge wins. */
448
+ /** See {@link GraphSearchHost.contentKeyEnds}. */
449
+ contentKeyEnds(prefix: Uint8Array, tail: Uint8Array): readonly number[] {
450
+ return keyEnds(this, prefix, tail);
451
+ }
452
+
377
453
  chooseNext(node: number): number | undefined {
378
454
  return chooseNext(this, node, this._edgeGuide);
379
455
  }
@@ -737,6 +813,55 @@ export class Mind implements MindContext {
737
813
  return decodeText(r.bytes);
738
814
  }
739
815
 
816
+ // ── Reading the trained memory back ─────────────────────────────────────
817
+
818
+ /** Which stored notes does this query REACH? BYTES in, BYTES out — this
819
+ * method has no notion of text or encoding; the text case is
820
+ * {@link searchCorpusText}, which is one caller of this.
821
+ *
822
+ * Exact content addressing through the machinery an answer already uses
823
+ * (see src/mind/corpus.ts): the query's recognised sites are the resolved
824
+ * subtrees, the climb goes up from the biggest, and a result is a context
825
+ * that carries a learnt continuation. Nothing is written and nothing is
826
+ * indexed. */
827
+ searchCorpus(queryBytes: Uint8Array, limit?: number): CorpusResult {
828
+ return searchCorpus(this, queryBytes, limit);
829
+ }
830
+
831
+ /** Browse real pairs. Deterministic: `from` is the caller's own offset in
832
+ * [0,1), so browsing twice with different offsets shows different notes
833
+ * without a random draw. */
834
+ sampleCorpus(limit?: number, from?: number): CorpusResult {
835
+ return sampleCorpus(this, limit, from);
836
+ }
837
+
838
+ /** The TEXT case of {@link searchCorpus}: encode, search, decode. The search
839
+ * itself exists once, in the byte layer above; only the rendering lives
840
+ * here, with the rest of this class's text modality. */
841
+ searchCorpusText(query: string, limit?: number): CorpusTextResult {
842
+ const result = this.searchCorpus(
843
+ new TextEncoder().encode(query),
844
+ limit,
845
+ );
846
+ return {
847
+ query,
848
+ pairs: result.pairs.map((p: CorpusPair): CorpusTextPair => ({
849
+ context: previewCorpusText(p.context),
850
+ continuation: previewCorpusText(p.continuation),
851
+ contextId: p.contextId,
852
+ continuationId: p.continuationId,
853
+ matchedBytes: p.matchedBytes,
854
+ contextTruncated: p.contextTruncated,
855
+ continuationTruncated: p.continuationTruncated,
856
+ })),
857
+ resolved: result.resolved,
858
+ reached: result.reached,
859
+ totalContexts: result.totalContexts,
860
+ browsed: result.browsed,
861
+ note: result.miss === "matched" ? undefined : CORPUS_NOTE[result.miss],
862
+ };
863
+ }
864
+
740
865
  // ── Conversation API ────────────────────────────────────────────────────
741
866
 
742
867
  /** Begin a new conversation, optionally restoring from a previously-saved
@@ -675,6 +675,13 @@ export interface MechanismResult {
675
675
  bytes: Uint8Array;
676
676
  accounted: Array<[number, number]>;
677
677
  moves: number;
678
+ /** WHAT THIS ANSWER SPEAKS FOR — the anchors it voices, and therefore the
679
+ * content the reasoner must not pivot back through. Declared by the
680
+ * mechanism about its OWN result, exactly like `accounted`/`unexplained`/
681
+ * `complete`: post-grounding honours the property and NEVER ASKS WHICH
682
+ * MECHANISM SET IT, so the market stays uniform. An EMPTY set is a real
683
+ * declaration — "this answer voices nothing" (recall) — and withholds
684
+ * nothing; omit the field and the pipeline re-recognises the answer. */
678
685
  used?: ReadonlySet<number>;
679
686
  unexplained: string;
680
687
  /** Explicit weight override. When absent, weight = moves + PASS·unaccounted. */