@hviana/sema 0.8.0 → 0.8.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (119) hide show
  1. package/AGENTS.md +22 -1
  2. package/DATASETS.md +1 -1
  3. package/dist/example/train_base/config.js +2 -2
  4. package/dist/example/train_base/corpora/massive.js +1 -1
  5. package/dist/example/train_base/readers.js +1 -1
  6. package/dist/src/config.d.ts +17 -0
  7. package/dist/src/config.js +18 -0
  8. package/dist/src/geometry.d.ts +10 -10
  9. package/dist/src/geometry.js +25 -24
  10. package/dist/src/meter.d.ts +29 -12
  11. package/dist/src/meter.js +58 -14
  12. package/dist/src/mind/attention.js +12 -12
  13. package/dist/src/mind/bridge.d.ts +8 -8
  14. package/dist/src/mind/bridge.js +33 -32
  15. package/dist/src/mind/corpus.d.ts +40 -0
  16. package/dist/src/mind/corpus.js +149 -0
  17. package/dist/src/mind/graph-search.d.ts +7 -8
  18. package/dist/src/mind/graph-search.js +244 -32
  19. package/dist/src/mind/index.d.ts +3 -1
  20. package/dist/src/mind/index.js +1 -0
  21. package/dist/src/mind/junction.d.ts +1 -1
  22. package/dist/src/mind/junction.js +8 -8
  23. package/dist/src/mind/learning.js +36 -35
  24. package/dist/src/mind/match.d.ts +8 -3
  25. package/dist/src/mind/match.js +156 -71
  26. package/dist/src/mind/mechanisms/cast.js +18 -2
  27. package/dist/src/mind/mechanisms/cover.js +19 -12
  28. package/dist/src/mind/mechanisms/prefix-completion.js +24 -24
  29. package/dist/src/mind/mechanisms/recall.js +38 -40
  30. package/dist/src/mind/mechanisms/reference.js +16 -16
  31. package/dist/src/mind/mind.d.ts +61 -7
  32. package/dist/src/mind/mind.js +72 -2
  33. package/dist/src/mind/pipeline-mechanism.d.ts +10 -8
  34. package/dist/src/mind/pipeline-mechanism.js +25 -21
  35. package/dist/src/mind/pipeline.d.ts +9 -9
  36. package/dist/src/mind/pipeline.js +49 -29
  37. package/dist/src/mind/primitives.d.ts +5 -5
  38. package/dist/src/mind/primitives.js +5 -5
  39. package/dist/src/mind/reasoning.d.ts +5 -1
  40. package/dist/src/mind/reasoning.js +54 -1
  41. package/dist/src/mind/recognition.d.ts +14 -13
  42. package/dist/src/mind/recognition.js +23 -23
  43. package/dist/src/mind/resonance.js +21 -21
  44. package/dist/src/mind/traverse.d.ts +54 -52
  45. package/dist/src/mind/traverse.js +83 -73
  46. package/dist/src/mind/types.d.ts +26 -4
  47. package/dist/src/store.d.ts +12 -12
  48. package/dist/src/store.js +12 -12
  49. package/docs/INDEX.md +2 -2
  50. package/docs/architecture/exact-vs-approximate.md +2 -1
  51. package/docs/architecture/fold-contract.md +1 -1
  52. package/docs/failures/tempting-but-wrong.md +33 -5
  53. package/docs/harness/gates.md +7 -7
  54. package/example/train_base/config.ts +2 -2
  55. package/example/train_base/corpora/massive.ts +1 -1
  56. package/example/train_base/readers.ts +1 -1
  57. package/jsr.json +1 -1
  58. package/package.json +1 -1
  59. package/src/config.ts +35 -0
  60. package/src/geometry.ts +25 -24
  61. package/src/meter.ts +61 -14
  62. package/src/mind/attention.ts +12 -12
  63. package/src/mind/bridge.ts +33 -32
  64. package/src/mind/corpus.ts +202 -0
  65. package/src/mind/graph-search.ts +261 -31
  66. package/src/mind/index.ts +8 -1
  67. package/src/mind/junction.ts +8 -8
  68. package/src/mind/learning.ts +36 -35
  69. package/src/mind/match.ts +163 -73
  70. package/src/mind/mechanisms/cast.ts +17 -1
  71. package/src/mind/mechanisms/cover.ts +18 -12
  72. package/src/mind/mechanisms/prefix-completion.ts +24 -24
  73. package/src/mind/mechanisms/recall.ts +38 -40
  74. package/src/mind/mechanisms/reference.ts +16 -16
  75. package/src/mind/mind.ts +129 -7
  76. package/src/mind/pipeline-mechanism.ts +25 -21
  77. package/src/mind/pipeline.ts +63 -38
  78. package/src/mind/primitives.ts +5 -5
  79. package/src/mind/reasoning.ts +55 -0
  80. package/src/mind/recognition.ts +23 -23
  81. package/src/mind/resonance.ts +21 -21
  82. package/src/mind/traverse.ts +83 -73
  83. package/src/mind/types.ts +30 -4
  84. package/src/store.ts +20 -20
  85. package/test/08-storage.test.mjs +1 -1
  86. package/test/100-complete-grounding-trace.test.mjs +109 -0
  87. package/test/101-alignment-gap-bound.test.mjs +106 -0
  88. package/test/102-production-composes-at-scale.test.mjs +110 -0
  89. package/test/103-alignment-gap-budget.test.mjs +89 -0
  90. package/test/104-composition-is-reported.test.mjs +90 -0
  91. package/test/105-derive-through-reports-its-refusal.test.mjs +113 -0
  92. package/test/106-the-join-fires.test.mjs +94 -0
  93. package/test/107-the-join-is-counted.test.mjs +81 -0
  94. package/test/108-the-join-chains.test.mjs +78 -0
  95. package/test/109-the-pivot-is-counted.test.mjs +60 -0
  96. package/test/110-the-reasoner-stops-when-the-question-is-answered.test.mjs +91 -0
  97. package/test/111-the-cover-assembly-is-counted.test.mjs +74 -0
  98. package/test/112-the-exploration-does-not-grow-with-the-hub.test.mjs +89 -0
  99. package/test/113-the-rationale-payload-is-bounded.test.mjs +84 -0
  100. package/test/114-alignment-budget-is-per-sweep.test.mjs +93 -0
  101. package/test/116-the-extension-is-gated-by-the-pipelines-own-remainder.test.mjs +100 -0
  102. package/test/117-corpus-search.test.mjs +171 -0
  103. package/test/14-scaling.test.mjs +10 -7
  104. package/test/35-prefix-edge.test.mjs +1 -1
  105. package/test/40-choosenext-scale-guard.test.mjs +16 -17
  106. package/test/56-bridge-identity-admission.test.mjs +6 -6
  107. package/test/70-prefix-completion.test.mjs +4 -3
  108. package/test/72-prefix-candidate-supply.test.mjs +3 -3
  109. package/test/73-scaffolding-only-bridge-abstains.test.mjs +6 -6
  110. package/test/75-multiturn-context-optimisation.test.mjs +5 -5
  111. package/test/76-reference-binding.test.mjs +6 -1
  112. package/test/84-composed-answer-honesty.test.mjs +5 -6
  113. package/test/88-dependency-footprint.test.mjs +1 -1
  114. package/test/89-completion-recursion.test.mjs +47 -19
  115. package/test/90-connector-read-cap.test.mjs +10 -8
  116. package/test/93-regime-prediction.test.mjs +10 -10
  117. package/test/94-cross-region-budget.test.mjs +2 -2
  118. package/test/95-wide-resonance-removed.test.mjs +8 -7
  119. package/test/96-bytes-walk-termination.test.mjs +3 -3
package/src/mind/match.ts CHANGED
@@ -38,7 +38,7 @@ import {
38
38
  identityBar,
39
39
  significanceBar,
40
40
  } from "../geometry.js";
41
- import { bytesEqual, indexOf } from "../bytes.js";
41
+ import { bytesEqual, indexOf, latin1 } from "../bytes.js";
42
42
  import type { MindContext } from "./types.js";
43
43
  import { chainReach, leafIdRun } from "./canonical.js";
44
44
  import { foldTree, gistOf, perceive, read, resolve } from "./primitives.js";
@@ -54,6 +54,7 @@ import {
54
54
  sharedReachMemo,
55
55
  } from "./traverse.js";
56
56
  import { recognise, segment } from "./recognition.js";
57
+ import { rItem } from "./trace.js";
57
58
  import type { Site } from "./graph-search.js";
58
59
 
59
60
  // ═══════════════════════════════════════════════════════════════════════════
@@ -319,12 +320,12 @@ export function alignGraded(
319
320
  //
320
321
  // "these bytes occupy a place the corpus keeps open" (bind).
321
322
  //
322
- // Both arrive as unaligned residue. That single missing distinction is why
323
- // the substitution bridge refuses on `attestedQ`, why the cover charges PASS
324
- // over a slot, and why CAST reads a filler as noise rather than as the
325
- // variable it is. The family below supplies it, and it lives HERE — not in
326
- // any mechanism — because it is the ordinary (matcher, projection, gate)
327
- // triple of §2.5 with its three parts in their proper places:
323
+ // Both arrive as unaligned residue. That single missing distinction is why the
324
+ // substitution bridge refuses on `attestedQ`, why the cover charges PASS over a
325
+ // slot, and why CAST reads a filler as noise rather than as the variable it is.
326
+ // The family below supplies it, and it lives HERE — not in any mechanism —
327
+ // because it is the ordinary (matcher, projection, gate) triple of
328
+ // match-project.md with its three parts in their proper places:
328
329
  //
329
330
  // matcher alignAround + contractGap + frameSlots — bytes only, no
330
331
  // projection, no licence. SAFE FOR EVERY CONSUMER: knowing a
@@ -360,9 +361,14 @@ export interface AlignGap {
360
361
 
361
362
  /** Extend a seed match (query offset qo ↔ candidate offset co) to its maximal
362
363
  * common run, then walk outward in both directions collecting further common
363
- * runs of at least W bytes across bounded mismatch gaps (each side ≤
364
- * chainReach). Returns the matched query spans and the mismatch pairs
365
- * between consecutive runs.
364
+ * runs of at least W bytes across mismatch gaps. Each gap's LENGTH is the
365
+ * pair's own extent (a gap cannot be longer than the bytes it spans) and the
366
+ * sweep's WORK is proportional to the bytes a run spans (the context's windows
367
+ * are indexed once, then the query's are walked) — the arity bound
368
+ * (`chainReach`) used to cap BOTH, and truncated every learned frame whose
369
+ * slot was longer. Each sweep owns its own budget, so an exhausted right
370
+ * sweep never starves the left one. Returns the matched query spans and the
371
+ * mismatch pairs between consecutive runs.
366
372
  *
367
373
  * This is the SEEDED aligner, distinct from {@link alignRuns}: that one finds
368
374
  * every run two structures share anywhere (a weave), this one reads two
@@ -382,7 +388,21 @@ export function alignAround(
382
388
  co: number,
383
389
  ): { matched: Array<[number, number]>; gaps: AlignGap[] } {
384
390
  const W = ctx.space.maxGroup;
385
- const reachCap = chainReach(W);
391
+ // THE GAP LENGTH IS THE PAIR'S OWN EXTENT; THE WORK IS BUDGETED.
392
+ //
393
+ // The sweep walks (queryGap, contextGap) pairs by ASCENDING total, so reaching
394
+ // a gap of size G costs about G²/2 pairs. Bounding the LENGTH by the write
395
+ // side's arity (`chainReach(W)` = 16) therefore truncated every learned frame
396
+ // whose slot is longer — measured: `bindReference` reported the cap at 18, 24,
397
+ // 30 and 36 bytes and `recall` answered with ANOTHER instance's filler — while
398
+ // removing the bound outright took the corpus-cost guard (test/89) from
399
+ // milliseconds to 68 seconds.
400
+ //
401
+ // Bounding the PAIRS keeps a call's cost constant however long the pair is,
402
+ // and the ascending order means an exhausted budget drops the FAR
403
+ // continuations and never the near ones — the same degradation recognition.ts
404
+ // documents for its canon budget. Length and work are different questions;
405
+ // this is the one place they were conflated.
386
406
  // Maximal run around the seed.
387
407
  let qs = qo, ss = co;
388
408
  while (qs > 0 && ss > 0 && q[qs - 1] === c[ss - 1]) {
@@ -396,8 +416,60 @@ export function alignAround(
396
416
  }
397
417
  const matched: Array<[number, number]> = [[qs, qe]];
398
418
  const gaps: AlignGap[] = [];
399
- // The next common run of ≥ W bytes past (qi, si), with each side's gap
400
- // bounded by chainReach; smallest total gap wins (nearest continuation).
419
+ // THE SWEEP IS STRUCTURAL, NOT ENUMERATIVE.
420
+ //
421
+ // The criterion is unchanged: the next common run, MINIMUM TOTAL GAP, ties to
422
+ // the smaller query gap. What changed is how it is found. Enumerating
423
+ // (queryGap, contextGap) pairs by ascending total reaches a run at total t in
424
+ // about t²/2 pairs — and that quadratic shape, not the reach, was the cost
425
+ // problem: capping the pairs dropped reach (a legitimate 24-byte slot stopped
426
+ // being found), while leaving them uncapped cost 68 seconds on the corpus
427
+ // guard. Neither is the answer, because the answer is the algorithm.
428
+ //
429
+ // The context's windows are indexed ONCE, for lengths 1..W — W being the
430
+ // geometry's own unit of composition, so nothing is chosen here. Each step
431
+ // then walks the query's windows outward from the anchor: for a given query
432
+ // gap the nearest context gap that continues a run is one O(1) lookup, and the
433
+ // walk stops the moment the query gap alone exceeds the best total already
434
+ // found. So the work is proportional to the bytes the run SPANS. No budget,
435
+ // no cap, no number: a long slot is reached, and its price is already the
436
+ // ladder's (its bytes are unaccounted, so the search pays PASS per byte).
437
+ const index: Array<Map<string, number[]>> = [];
438
+ for (let len = 1; len <= W; len++) {
439
+ const m = new Map<string, number[]>();
440
+ for (let o = 0; o + len <= c.length; o++) {
441
+ const key = latin1(c.subarray(o, o + len));
442
+ const at = m.get(key);
443
+ if (at === undefined) m.set(key, [o]);
444
+ else at.push(o);
445
+ }
446
+ index.push(m);
447
+ }
448
+ /** Smallest listed offset at or after `from`, or -1. */
449
+ const fromAt = (list: number[], from: number): number => {
450
+ let lo = 0, hi = list.length - 1, best = -1;
451
+ while (lo <= hi) {
452
+ const mid = (lo + hi) >> 1;
453
+ if (list[mid] >= from) {
454
+ best = list[mid];
455
+ hi = mid - 1;
456
+ } else lo = mid + 1;
457
+ }
458
+ return best;
459
+ };
460
+ /** Largest listed offset at or before `to`, or -1. */
461
+ const toAt = (list: number[], to: number): number => {
462
+ let lo = 0, hi = list.length - 1, best = -1;
463
+ while (lo <= hi) {
464
+ const mid = (lo + hi) >> 1;
465
+ if (list[mid] <= to) {
466
+ best = list[mid];
467
+ lo = mid + 1;
468
+ } else hi = mid - 1;
469
+ }
470
+ return best;
471
+ };
472
+ /** Length of the common run STARTING at (qi, si). */
401
473
  const runLenAt = (qi: number, si: number): number => {
402
474
  let n = 0;
403
475
  while (qi + n < q.length && si + n < c.length && q[qi + n] === c[si + n]) {
@@ -405,63 +477,80 @@ export function alignAround(
405
477
  }
406
478
  return n;
407
479
  };
408
- // RIGHT sweep.
409
- let qi = qe, si = se;
410
- for (;;) {
411
- let found = false;
412
- for (let total = 1; total <= 2 * reachCap && !found; total++) {
413
- for (let gq = 0; gq <= Math.min(total, reachCap); gq++) {
414
- const gs = total - gq;
415
- if (gs > reachCap) continue;
416
- if (qi + gq >= q.length || si + gs >= c.length) continue;
417
- const n = runLenAt(qi + gq, si + gs);
418
- if (n >= W || qi + gq + n === q.length) {
419
- if (n === 0) continue;
420
- if (gq > 0 || gs > 0) {
421
- gaps.push({ qs: qi, qe: qi + gq, cs: si, ce: si + gs });
480
+ /** Length of the common run ENDING at (qi, si). */
481
+ const runLenBefore = (qi: number, si: number): number => {
482
+ let n = 0;
483
+ while (n < qi && n < si && q[qi - 1 - n] === c[si - 1 - n]) n++;
484
+ return n;
485
+ };
486
+ /** The next run outward from an anchor, or null when the bytes run out. */
487
+ const nextRun = (
488
+ qi: number,
489
+ si: number,
490
+ forward: boolean,
491
+ ): { gq: number; gs: number; n: number } | null => {
492
+ const qLim = forward ? q.length - qi : qi;
493
+ let best: { gq: number; gs: number; n: number } | null = null;
494
+ for (let gq = 0; gq < qLim; gq++) {
495
+ // No later query gap can beat a total already found.
496
+ if (best !== null && gq > best.gq + best.gs) break;
497
+ const left = qLim - gq;
498
+ // A run of >= W bytes, or — when the query itself ends inside one window —
499
+ // the run that REACHES that end. Exactly the acceptance the sweep had.
500
+ const lens = left >= W ? [W] : [left];
501
+ for (const len of lens) {
502
+ const key = latin1(
503
+ q.subarray(
504
+ forward ? qi + gq : qi - gq - len,
505
+ forward ? qi + gq + len : qi - gq,
506
+ ),
507
+ );
508
+ const list = index[len - 1].get(key);
509
+ if (list === undefined) continue;
510
+ const o = forward ? fromAt(list, si) : toAt(list, si - len);
511
+ if (o < 0) continue;
512
+ const n = forward
513
+ ? runLenAt(qi + gq, o)
514
+ : runLenBefore(qi - gq, o + len);
515
+ if (n < 1) continue;
516
+ if (
517
+ forward ? n >= W || qi + gq + n === q.length : n >= W || n === qi - gq
518
+ ) {
519
+ const gs = forward ? o - si : si - len - o;
520
+ if (best === null || gq + gs < best.gq + best.gs) {
521
+ best = { gq, gs, n };
422
522
  }
423
- matched.push([qi + gq, qi + gq + n]);
424
- qi = qi + gq + n;
425
- si = si + gs + n;
426
- found = true;
427
523
  break;
428
524
  }
429
525
  }
430
526
  }
431
- if (!found) break;
527
+ return best;
528
+ };
529
+ // RIGHT sweep.
530
+ let qi = qe, si = se;
531
+ for (;;) {
532
+ const step = nextRun(qi, si, true);
533
+ if (step === null) break;
534
+ if (step.gq > 0 || step.gs > 0) {
535
+ gaps.push({ qs: qi, qe: qi + step.gq, cs: si, ce: si + step.gs });
536
+ }
537
+ matched.push([qi + step.gq, qi + step.gq + step.n]);
538
+ qi = qi + step.gq + step.n;
539
+ si = si + step.gs + step.n;
432
540
  }
433
- // LEFT sweep (mirror).
541
+ // LEFT sweep (mirror): an independent walk, so an exhausted right side can
542
+ // never starve it (pinned by test/114).
434
543
  qi = qs;
435
544
  si = ss;
436
545
  for (;;) {
437
- let found = false;
438
- for (let total = 1; total <= 2 * reachCap && !found; total++) {
439
- for (let gq = 0; gq <= Math.min(total, reachCap); gq++) {
440
- const gs = total - gq;
441
- if (gs > reachCap) continue;
442
- if (qi - gq <= 0 || si - gs <= 0) continue;
443
- // Run ENDING at (qi - gq, si - gs).
444
- let n = 0;
445
- while (
446
- n < qi - gq && n < si - gs &&
447
- q[qi - gq - 1 - n] === c[si - gs - 1 - n]
448
- ) {
449
- n++;
450
- }
451
- if (n >= W || n === qi - gq) {
452
- if (n === 0) continue;
453
- if (gq > 0 || gs > 0) {
454
- gaps.push({ qs: qi - gq, qe: qi, cs: si - gs, ce: si });
455
- }
456
- matched.push([qi - gq - n, qi - gq]);
457
- qi = qi - gq - n;
458
- si = si - gs - n;
459
- found = true;
460
- break;
461
- }
462
- }
546
+ const step = nextRun(qi, si, false);
547
+ if (step === null) break;
548
+ if (step.gq > 0 || step.gs > 0) {
549
+ gaps.push({ qs: qi - step.gq, qe: qi, cs: si - step.gs, ce: si });
463
550
  }
464
- if (!found) break;
551
+ matched.push([qi - step.gq - step.n, qi - step.gq]);
552
+ qi = qi - step.gq - step.n;
553
+ si = si - step.gs - step.n;
465
554
  }
466
555
  return { matched, gaps };
467
556
  }
@@ -1168,24 +1257,25 @@ export async function project(
1168
1257
 
1169
1258
  // ── The span-shape family ───────────────────────────────────────────────────
1170
1259
  //
1171
- // "Is this answer drawn from this context?" has TWO formally distinct
1172
- // readings, and the pair plus the anchor classifier built on them are SHARED
1173
- // machinery — extraction proposes span-shaped exemplars with them, the
1174
- // shared `Precomputed.spanShapedOf` container computes them, and fusion
1175
- // (reasoning.ts) gates on the strict one. They lived inside
1176
- // mechanisms/extraction.ts, so `pipeline-mechanism.ts` and `reasoning.ts`
1177
- // both had to import back OUT of a specific mechanism — an inversion the
1178
- // mechanism market forbids (AGENTS §2.6: the shared contract may not depend
1179
- // on any one mechanism; §2.5: a shared matcher belongs to this family, never
1180
- // to a mechanism's private helpers). Deleting extraction must not break the
1181
- // shared container, so they live here.
1260
+ // "Is this answer drawn from this context?" has TWO formally distinct readings,
1261
+ // and the pair plus the anchor classifier built on them are SHARED machinery —
1262
+ // extraction proposes span-shaped exemplars with them, the shared
1263
+ // `Precomputed.spanShapedOf` container computes them, and fusion (reasoning.ts)
1264
+ // gates on the strict one. They lived inside mechanisms/extraction.ts, so
1265
+ // `pipeline-mechanism.ts` and `reasoning.ts` both had to import back OUT of a
1266
+ // specific mechanism — an inversion the mechanism market forbids: the shared
1267
+ // contract may not depend on any one mechanism (mechanism-market.md), and a
1268
+ // shared matcher belongs to this family (match-project.md), never to a
1269
+ // mechanism's private helpers. Deleting extraction must not break the shared
1270
+ // container, so they live here.
1182
1271
  //
1183
1272
  // • isSpanShaped — the OPEN reading (sparse in-order embedding).
1184
1273
  // • containsSpan — the STRICT reading (contiguous run or resolved node).
1185
1274
  // • skillExemplar — classify one anchor into (context, answer) using them.
1186
1275
  //
1187
- // The two readings are NOT interchangeable; AGENTS §2.5 pins the distinction
1188
- // and each function's own doc states what breaks if it is substituted.
1276
+ // The two readings are NOT interchangeable; match-project.md pins the
1277
+ // distinction and each function's own doc states what breaks if it is
1278
+ // substituted.
1189
1279
 
1190
1280
  /** Check whether an anchor is a span-shaped skill exemplar: it represents a
1191
1281
  * fact whose context and answer together form a span-in-context pattern.
@@ -615,7 +615,23 @@ export async function counterfactualTransfer(
615
615
  fwd !== null && indexOf(answer, fwd, 0) < 0 &&
616
616
  !restatesQuery(query, fwd)
617
617
  ) {
618
- answer = concat2(answer, fwd);
618
+ // THROUGH THE SHARED JOINER, not a bare concatenation.
619
+ //
620
+ // `joinWithBridge` is the composition step every out-of-search assembly
621
+ // shares (multi-topic fusion, CAST's substitution and comparison): it
622
+ // asks the corpus for a learnt connector between the pieces and, on a
623
+ // miss, joins them BARE **and says so** — the `bridgeMiss` step (see
624
+ // resonance.ts). This site bypassed it, and that is the whole of the
625
+ // gluing the study measured: `"Steel is hard"` + `"wet"` came back as
626
+ // `"hardwet"`, `"eva director father"` + `"The father of…"` as
627
+ // `"fatherThe"` — compositions no rationale could show, because the one
628
+ // step that made them left no trace.
629
+ //
630
+ // Routing it through the shared joiner is the instrumentation fix that
631
+ // comes first: a bare join stays possible (the house rule is "joined
632
+ // bare, never silent") but it is now VISIBLE, and an attested connector
633
+ // is used when the corpus has one.
634
+ answer = await joinWithBridge(ctx, answer, fwd);
619
635
  }
620
636
  ctx.trace?.step(
621
637
  "projectCounterfactual",
@@ -4,7 +4,7 @@
4
4
  // Cover consumes recognition directly (its axioms are the query's own
5
5
  // decomposition) plus the computed spans any parse()-bearing mechanism
6
6
  // contributed: computed spans MASK colliding recognised sites and enter the
7
- // search at zero cost ("computation always wins", §16.3) — which is also why
7
+ // search at zero cost ("computation always wins", alu.md) — which is also why
8
8
  // cover runs FIRST in defaultMechanisms: a computed-backed cover becomes a
9
9
  // near-zero-cost incumbent that prunes the other mechanisms through the
10
10
  // ordinary admissible-floor check, with no extension special-case anywhere.
@@ -75,28 +75,30 @@ export async function resolveConnectors(
75
75
  if (query === undefined || ctx.answeredSpans.length === 0) return true;
76
76
  const continuations = ctx.store.nextFirst(s.payload, hubBound(ctx));
77
77
  return !continuations.some((answer) => {
78
- // PREFIX-CAPPED (AGENTS §2.8): a candidate longer than the query cannot
79
- // occur INSIDE it, so read one byte past the query's length — enough to
80
- // detect the overflow — and reject without reconstructing the rest.
81
- // The `+ 1` is what makes the test exact rather than a truncation: a
82
- // result of exactly `query.length + 1` bytes is known to be too long,
83
- // and anything shorter is the candidate's COMPLETE content, so the
84
- // substring test below is the same test as before. (The same overflow
85
- // probe bridge.ts:256 already uses.)
78
+ // PREFIX-CAPPED (bounded-reads.md): a candidate longer than the query
79
+ // cannot occur INSIDE it, so read one byte past the query's length —
80
+ // enough to detect the overflow — and reject without reconstructing the
81
+ // rest. The `+ 1` is what makes the test exact rather than a
82
+ // truncation: a result of exactly `query.length + 1` bytes is known to
83
+ // be too long, and anything shorter is the candidate's COMPLETE
84
+ // content, so the substring test below is the same test as before. (The
85
+ // same overflow probe bridge.ts:256 already uses.)
86
86
  //
87
87
  // This loop runs up to hubBound(ctx) = √N reads PER SITE, and only on a
88
88
  // multi-turn response — `answeredSpans` is empty for a plain respond(),
89
- // so the probe does not execute there. The cap cannot reduce the read
89
+ // so the probe does not execute there. The cap cannot reduce the read
90
90
  // COUNT — only a semantic change to the "already answered" test could —
91
91
  // but it bounds each read by the query instead of by the corpus, which
92
- // is what §2.8 asks for and what rescues a SHORT query: at 3 bytes this
93
- // reads 4 bytes per candidate instead of the ~231 it averaged before.
92
+ // is what bounded-reads.md asks for and what rescues a SHORT query: at
93
+ // 3 bytes this reads 4 bytes per candidate instead of the ~231 it
94
+ // averaged before.
94
95
  const bytes = read(ctx, answer, query.length + 1);
95
96
  return bytes.length <= query.length && indexOf(query, bytes, 0) >= 0;
96
97
  });
97
98
  });
98
99
  const bridgePair = async (l: number, r: number) => {
99
100
  if (l === r || links.has(l + "," + r)) return;
101
+ if (ctx.meter) ctx.meter.coverBridges++;
100
102
  const link = await bridge(ctx, read(ctx, l), read(ctx, r));
101
103
  if (link !== null) links.set(l + "," + r, link);
102
104
  };
@@ -133,6 +135,10 @@ export async function resolveConnectors(
133
135
  // plus one W-quantum of glue per joint — pass that allowance so the
134
136
  // bridge's phrase-scale cap admits the whole learnt run.
135
137
  const allowance = middleBytes + (m + 1) * W;
138
+ if (ctx.meter) {
139
+ ctx.meter.coverBridges++;
140
+ ctx.meter.coverAllowanceBytes += allowance;
141
+ }
136
142
  const interior = await bridge(
137
143
  ctx,
138
144
  first.bytes,
@@ -1,15 +1,15 @@
1
1
  // mechanisms/prefix-completion.ts — Grounding a query that IS the opening of a
2
2
  // trained form (Grounding V).
3
3
  //
4
- // A MECHANISM, NOT A TIER. This used to run inside recall's refusal path, in
5
- // a fixed if-chain that first-match-wins — the shape CAST was refactored away
6
- // from, where placement rather than the cost ladder decided. Its claim is
4
+ // A MECHANISM, NOT A TIER. This used to run inside recall's refusal path, in a
5
+ // fixed if-chain that first-match-wins — the shape CAST was refactored away
6
+ // from, where placement rather than the cost ladder decided. Its claim is
7
7
  // maximal (every query byte literally matched, from offset zero, against a
8
8
  // trained form) at one STEP, so as a market candidate it competes honestly and
9
- // the decider weighs it like everything else. It is registered LAST: recall's
9
+ // the decider weighs it like everything else. It is registered LAST: recall's
10
10
  // exact self-match makes an IDENTITY claim about the query while this makes a
11
- // CONTAINMENT one, and on an exact grade tie the identity claim is the
12
- // stronger evidence — the same ordering §2.3's ladders use.
11
+ // CONTAINMENT one, and on an exact grade tie the identity claim is the stronger
12
+ // evidence — the same ordering exact-vs-approximate.md's ladders use.
13
13
  //
14
14
  // Its SUPPLY moved too, and further: `formsOpenedBy` (traverse.ts) answers a
15
15
  // question about the STORE — "which trained forms does this byte run open?" —
@@ -49,12 +49,12 @@
49
49
  //
50
50
  // So this is a RETRIEVABILITY gap, not a semantic one, and the ANN is the wrong
51
51
  // instrument for it: a proper prefix's gist cannot rank its own continuation.
52
- // The repair is CONTENT-ADDRESSED (§2.3) — `formsOpenedBy` (traverse.ts) reads
53
- // the leaf-id WINDOW index the write side already maintains and answers "which
54
- // trained forms does this byte run open?" in a bounded √N walk. That is this
55
- // mechanism's first supply. The response's memoised top-k `resonance()` is the
56
- // second, for prefixes long enough that the gist still ranks the form; it is
57
- // read, never re-issued.
52
+ // The repair is CONTENT-ADDRESSED (exact-vs-approximate.md) — `formsOpenedBy`
53
+ // (traverse.ts) reads the leaf-id WINDOW index the write side already maintains
54
+ // and answers "which trained forms does this byte run open?" in a bounded √N
55
+ // walk. That is this mechanism's first supply. The response's memoised top-k
56
+ // `resonance()` is the second, for prefixes long enough that the gist still
57
+ // ranks the form; it is read, never re-issued.
58
58
  //
59
59
  // AN EXHAUSTIVE ANN LIST IS NOT A SUPPLY HERE, AND WAS REMOVED. This tier once
60
60
  // read `Precomputed.wideResonance()` — a full-index `resonate(guide, √N,
@@ -273,20 +273,20 @@ export const prefixMechanism: PipelineMechanism = {
273
273
  return STEP;
274
274
  },
275
275
  async run(ctx, query, pre) {
276
- // ONE SUPPLY PASS, not a two-tier `??`. The window index (exact,
276
+ // ONE SUPPLY PASS, not a two-tier `??`. The window index (exact,
277
277
  // content-addressed) and the response's memoised top-k (approximate) are
278
- // concatenated and the three guards decide ONCE over the union. A
278
+ // concatenated and the three guards decide ONCE over the union. A
279
279
  // first-then-fallback chain would let the APPROXIMATE tier override the
280
- // EXACT one (§2.3): when formsOpenedBy finds two continuations, guard 3
281
- // returns null and the fallback re-runs the guards on resonance's top-k
282
- // alone — which, seeing only one of the two forms, would voice it. That is
283
- // precisely the disagreement-suppression guard 3 exists to prevent, and it
284
- // is the exact tier's ambiguity being washed away by the approximate tier.
285
- // Evaluating the union means a disagreement the window index saw can never
286
- // be hidden by what the ANN happens to rank. The ANN read is the
287
- // response's ONE memoised top-k (§2.11), already paid by recall's refusal
288
- // path on the queries where this mechanism fires, so reading it here is not
289
- // a second index scan.
280
+ // EXACT one (exact-vs-approximate.md): when formsOpenedBy finds two
281
+ // continuations, guard 3 returns null and the fallback re-runs the guards
282
+ // on resonance's top-k alone — which, seeing only one of the two forms,
283
+ // would voice it. That is precisely the disagreement-suppression guard 3
284
+ // exists to prevent, and it is the exact tier's ambiguity being washed away
285
+ // by the approximate tier. Evaluating the union means a disagreement the
286
+ // window index saw can never be hidden by what the ANN happens to rank. The
287
+ // ANN read is the response's ONE memoised top-k (memoization.md), already
288
+ // paid by recall's refusal path on the queries where this mechanism fires,
289
+ // so reading it here is not a second index scan.
290
290
  const ids = [
291
291
  ...formsOpenedBy(ctx, query),
292
292
  ...(await pre.resonance()).map((h) => h.id),
@@ -236,24 +236,22 @@ export async function recallByResonance(
236
236
  }
237
237
  }
238
238
 
239
- // The query-relative grounding fraction, shared by tiers 2–4 — gated on
240
- // the FRACTION OF THE QUERY the grounding explains, not the raw cosine.
241
- // Root gists are unit vectors, but their magnitudes are recoverable from
242
- // the byte lengths (‖·‖ = √len under the linear fold):
243
- // cos = shared/√(lenQ·lenG), so shared/lenQ = cos·√(lenG/lenQ).
244
- // The raw cosine punished honest containment — a query fully inside a
245
- // longer grounded answer scored √(lenQ/lenG) and was refused — and let a
246
- // long answer sharing only scaffolding pass; the query-relative fraction
247
- // measures exactly what the reach bar means: how much of THE QUERY the
248
- // store accounts for.
249
- // Chance similarity survives the length conversion AMPLIFIED: the same
250
- // √(lenG/lenQ) factor that converts an honest shared fraction into a
251
- // query-relative one multiplies the estimator/chance floor too, so a long
252
- // stored form (√(lenG/lenQ) ≈ 10 at 100×) lifted a noise-level cosine past
253
- // the reach bar and grounded pure gibberish (observed). Only the
254
- // ABOVE-CHANCE part of the similarity is evidence of shared content —
255
- // subtract the significance bar (3/√D, §8.3) before converting. Derived
256
- // from the existing bars; never tuned.
239
+ // The query-relative grounding fraction, shared by tiers 2–4 — gated on the
240
+ // FRACTION OF THE QUERY the grounding explains, not the raw cosine. Root
241
+ // gists are unit vectors, but their magnitudes are recoverable from the byte
242
+ // lengths (‖·‖ = √len under the linear fold): cos = shared/√(lenQ·lenG), so
243
+ // shared/lenQ = cos·√(lenG/lenQ). The raw cosine punished honest containment
244
+ // — a query fully inside a longer grounded answer scored √(lenQ/lenG) and was
245
+ // refused — and let a long answer sharing only scaffolding pass; the
246
+ // query-relative fraction measures exactly what the reach bar means: how much
247
+ // of THE QUERY the store accounts for. Chance similarity survives the length
248
+ // conversion AMPLIFIED: the same √(lenG/lenQ) factor that converts an honest
249
+ // shared fraction into a query-relative one multiplies the estimator/chance
250
+ // floor too, so a long stored form (√(lenG/lenQ) ≈ 10 at 100×) lifted a
251
+ // noise-level cosine past the reach bar and grounded pure gibberish
252
+ // (observed). Only the ABOVE-CHANCE part of the similarity is evidence of
253
+ // shared content — subtract the significance bar (3/√D, thresholds.md) before
254
+ // converting. Derived from the existing bars; never tuned.
257
255
  const sig = significanceBar(ctx.store.D);
258
256
  const reach = reachThreshold(ctx.space.maxGroup);
259
257
  const fracOfQuery = (cos: number, otherLen: number): number =>
@@ -411,14 +409,14 @@ export async function recallByResonance(
411
409
  }
412
410
  }
413
411
  }
414
- // 3b. Corroborated-substitution bridge — refusal-path only (bridge.ts).
415
- // The bridge's proposal source is the response's ONE top-k read — the same
416
- // list recall already ranked above — never an exhaustive √N scan. The
417
- // bridge's own candidate cap is 2·recallQueryK, so top-k proposals are
418
- // exactly the budget it can consume, and every proposal is byte-verified
419
- // downstream (§2.3). Reuse the memoised `resonance()`; scanning every IVF
420
- // cluster here once made every honest refusal cost hundreds of ms regardless
421
- // of k.
412
+ // 3b. Corroborated-substitution bridge — refusal-path only (bridge.ts). The
413
+ // bridge's proposal source is the response's ONE top-k read — the same list
414
+ // recall already ranked above — never an exhaustive √N scan. The bridge's own
415
+ // candidate cap is 2·recallQueryK, so top-k proposals are exactly the budget
416
+ // it can consume, and every proposal is byte-verified downstream
417
+ // (exact-vs-approximate.md). Reuse the memoised `resonance()`; scanning every
418
+ // IVF cluster here once made every honest refusal cost hundreds of ms
419
+ // regardless of k.
422
420
  const wideIds = async () => (await pre.resonance()).map((h) => h.id);
423
421
 
424
422
  // Every gist-based tier has failed; before refusing, align the query
@@ -471,12 +469,12 @@ export async function recallByResonance(
471
469
  // prefixCompletion runs a few lines below and carries the three guards
472
470
  // this tier lacks — unreadable-continuation veto, sub-quantum
473
471
  // continuation, and UNIQUENESS (distinct continuations ⇒ refuse), which
474
- // is exactly what 4,300 competing values must trip. So this is not a
475
- // new rule and not a new threshold: it is deferring a prefix decision to
476
- // the tier that owns it (§2.5, one factored machinery). Byte-strict on
477
- // purpose — a candidate differing by case or punctuation ("what is the
478
- // capital of france" → "What is the capital of France?") is NOT a byte
479
- // prefix, keeps grounding here, and is unaffected.
472
+ // is exactly what 4,300 competing values must trip. So this is not a new
473
+ // rule and not a new threshold: it is deferring a prefix decision to the
474
+ // tier that owns it (match-project.md, one factored machinery).
475
+ // Byte-strict on purpose — a candidate differing by case or punctuation
476
+ // ("what is the capital of france" → "What is the capital of France?") is
477
+ // NOT a byte prefix, keeps grounding here, and is unaffected.
480
478
  const strictPrefix = g !== null &&
481
479
  cBytes.length > query.length &&
482
480
  indexOf(cBytes, query, 0) === 0;
@@ -542,15 +540,15 @@ export async function recallByResonance(
542
540
  }
543
541
  }
544
542
 
545
- // The refusal/echo decision. The echo returns a stored form's bytes AS
546
- // the answer — a near-identity claim about the query — and identity-grade
543
+ // The refusal/echo decision. The echo returns a stored form's bytes AS the
544
+ // answer — a near-identity claim about the query — and identity-grade
547
545
  // decisions are never made on an estimated score ("approximate scores may
548
- // rank and propose; they may never decide", §6.2): the RaBitQ estimate
549
- // overshooting the reach bar echoed a WRONG-entity neighbour ("capital of
550
- // Zamunda?" echoed the Armenia fact, observed). The bytes are read
551
- // anyway to be echoed, so the decision uses their EXACT fold: one river
552
- // fold of the top hit, measured in the same query-relative,
553
- // chance-corrected units as the tier above.
546
+ // rank and propose; they may never decide", exact-vs-approximate.md): the
547
+ // RaBitQ estimate overshooting the reach bar echoed a WRONG-entity neighbour
548
+ // ("capital of Zamunda?" echoed the Armenia fact, observed). The bytes are
549
+ // read anyway to be echoed, so the decision uses their EXACT fold: one river
550
+ // fold of the top hit, measured in the same query-relative, chance-corrected
551
+ // units as the tier above.
554
552
  const topBytes = read(ctx, top.id);
555
553
  const exact = topBytes.length > 0
556
554
  ? cosine(queryGist, gistOf(ctx, topBytes))