@hviana/sema 0.8.0 → 0.8.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (119) hide show
  1. package/AGENTS.md +22 -1
  2. package/DATASETS.md +1 -1
  3. package/dist/example/train_base/config.js +2 -2
  4. package/dist/example/train_base/corpora/massive.js +1 -1
  5. package/dist/example/train_base/readers.js +1 -1
  6. package/dist/src/config.d.ts +17 -0
  7. package/dist/src/config.js +18 -0
  8. package/dist/src/geometry.d.ts +10 -10
  9. package/dist/src/geometry.js +25 -24
  10. package/dist/src/meter.d.ts +29 -12
  11. package/dist/src/meter.js +58 -14
  12. package/dist/src/mind/attention.js +12 -12
  13. package/dist/src/mind/bridge.d.ts +8 -8
  14. package/dist/src/mind/bridge.js +33 -32
  15. package/dist/src/mind/corpus.d.ts +40 -0
  16. package/dist/src/mind/corpus.js +149 -0
  17. package/dist/src/mind/graph-search.d.ts +7 -8
  18. package/dist/src/mind/graph-search.js +244 -32
  19. package/dist/src/mind/index.d.ts +3 -1
  20. package/dist/src/mind/index.js +1 -0
  21. package/dist/src/mind/junction.d.ts +1 -1
  22. package/dist/src/mind/junction.js +8 -8
  23. package/dist/src/mind/learning.js +36 -35
  24. package/dist/src/mind/match.d.ts +8 -3
  25. package/dist/src/mind/match.js +156 -71
  26. package/dist/src/mind/mechanisms/cast.js +18 -2
  27. package/dist/src/mind/mechanisms/cover.js +19 -12
  28. package/dist/src/mind/mechanisms/prefix-completion.js +24 -24
  29. package/dist/src/mind/mechanisms/recall.js +38 -40
  30. package/dist/src/mind/mechanisms/reference.js +16 -16
  31. package/dist/src/mind/mind.d.ts +61 -7
  32. package/dist/src/mind/mind.js +72 -2
  33. package/dist/src/mind/pipeline-mechanism.d.ts +10 -8
  34. package/dist/src/mind/pipeline-mechanism.js +25 -21
  35. package/dist/src/mind/pipeline.d.ts +9 -9
  36. package/dist/src/mind/pipeline.js +49 -29
  37. package/dist/src/mind/primitives.d.ts +5 -5
  38. package/dist/src/mind/primitives.js +5 -5
  39. package/dist/src/mind/reasoning.d.ts +5 -1
  40. package/dist/src/mind/reasoning.js +54 -1
  41. package/dist/src/mind/recognition.d.ts +14 -13
  42. package/dist/src/mind/recognition.js +23 -23
  43. package/dist/src/mind/resonance.js +21 -21
  44. package/dist/src/mind/traverse.d.ts +54 -52
  45. package/dist/src/mind/traverse.js +83 -73
  46. package/dist/src/mind/types.d.ts +26 -4
  47. package/dist/src/store.d.ts +12 -12
  48. package/dist/src/store.js +12 -12
  49. package/docs/INDEX.md +2 -2
  50. package/docs/architecture/exact-vs-approximate.md +2 -1
  51. package/docs/architecture/fold-contract.md +1 -1
  52. package/docs/failures/tempting-but-wrong.md +33 -5
  53. package/docs/harness/gates.md +7 -7
  54. package/example/train_base/config.ts +2 -2
  55. package/example/train_base/corpora/massive.ts +1 -1
  56. package/example/train_base/readers.ts +1 -1
  57. package/jsr.json +1 -1
  58. package/package.json +1 -1
  59. package/src/config.ts +35 -0
  60. package/src/geometry.ts +25 -24
  61. package/src/meter.ts +61 -14
  62. package/src/mind/attention.ts +12 -12
  63. package/src/mind/bridge.ts +33 -32
  64. package/src/mind/corpus.ts +202 -0
  65. package/src/mind/graph-search.ts +261 -31
  66. package/src/mind/index.ts +8 -1
  67. package/src/mind/junction.ts +8 -8
  68. package/src/mind/learning.ts +36 -35
  69. package/src/mind/match.ts +163 -73
  70. package/src/mind/mechanisms/cast.ts +17 -1
  71. package/src/mind/mechanisms/cover.ts +18 -12
  72. package/src/mind/mechanisms/prefix-completion.ts +24 -24
  73. package/src/mind/mechanisms/recall.ts +38 -40
  74. package/src/mind/mechanisms/reference.ts +16 -16
  75. package/src/mind/mind.ts +129 -7
  76. package/src/mind/pipeline-mechanism.ts +25 -21
  77. package/src/mind/pipeline.ts +63 -38
  78. package/src/mind/primitives.ts +5 -5
  79. package/src/mind/reasoning.ts +55 -0
  80. package/src/mind/recognition.ts +23 -23
  81. package/src/mind/resonance.ts +21 -21
  82. package/src/mind/traverse.ts +83 -73
  83. package/src/mind/types.ts +30 -4
  84. package/src/store.ts +20 -20
  85. package/test/08-storage.test.mjs +1 -1
  86. package/test/100-complete-grounding-trace.test.mjs +109 -0
  87. package/test/101-alignment-gap-bound.test.mjs +106 -0
  88. package/test/102-production-composes-at-scale.test.mjs +110 -0
  89. package/test/103-alignment-gap-budget.test.mjs +89 -0
  90. package/test/104-composition-is-reported.test.mjs +90 -0
  91. package/test/105-derive-through-reports-its-refusal.test.mjs +113 -0
  92. package/test/106-the-join-fires.test.mjs +94 -0
  93. package/test/107-the-join-is-counted.test.mjs +81 -0
  94. package/test/108-the-join-chains.test.mjs +78 -0
  95. package/test/109-the-pivot-is-counted.test.mjs +60 -0
  96. package/test/110-the-reasoner-stops-when-the-question-is-answered.test.mjs +91 -0
  97. package/test/111-the-cover-assembly-is-counted.test.mjs +74 -0
  98. package/test/112-the-exploration-does-not-grow-with-the-hub.test.mjs +89 -0
  99. package/test/113-the-rationale-payload-is-bounded.test.mjs +84 -0
  100. package/test/114-alignment-budget-is-per-sweep.test.mjs +93 -0
  101. package/test/116-the-extension-is-gated-by-the-pipelines-own-remainder.test.mjs +100 -0
  102. package/test/117-corpus-search.test.mjs +171 -0
  103. package/test/14-scaling.test.mjs +10 -7
  104. package/test/35-prefix-edge.test.mjs +1 -1
  105. package/test/40-choosenext-scale-guard.test.mjs +16 -17
  106. package/test/56-bridge-identity-admission.test.mjs +6 -6
  107. package/test/70-prefix-completion.test.mjs +4 -3
  108. package/test/72-prefix-candidate-supply.test.mjs +3 -3
  109. package/test/73-scaffolding-only-bridge-abstains.test.mjs +6 -6
  110. package/test/75-multiturn-context-optimisation.test.mjs +5 -5
  111. package/test/76-reference-binding.test.mjs +6 -1
  112. package/test/84-composed-answer-honesty.test.mjs +5 -6
  113. package/test/88-dependency-footprint.test.mjs +1 -1
  114. package/test/89-completion-recursion.test.mjs +47 -19
  115. package/test/90-connector-read-cap.test.mjs +10 -8
  116. package/test/93-regime-prediction.test.mjs +10 -10
  117. package/test/94-cross-region-budget.test.mjs +2 -2
  118. package/test/95-wide-resonance-removed.test.mjs +8 -7
  119. package/test/96-bytes-walk-termination.test.mjs +3 -3
@@ -268,7 +268,8 @@ function constituentSketch(ctx, id, k) {
268
268
  pool.push(g);
269
269
  }
270
270
  }
271
- // Bottom-k by identity, then by id so ties are corpus-determined (§2.1).
271
+ // Bottom-k by identity, then by id so ties are corpus-determined
272
+ // (determinism.md).
272
273
  pool.sort((a, b) => (unitPriority(a) - unitPriority(b)) || (a - b));
273
274
  const seen = new Set();
274
275
  out = [];
@@ -331,15 +332,15 @@ function constituentSketch(ctx, id, k) {
331
332
  * terms unique to that partner, which dilute but never mislead; the shared
332
333
  * units contribute the signal.
333
334
  *
334
- * HUBS ARE THE ONE EXCLUSION, read LIMITed as `parentsFirst(n, bound+1)` —
335
- * the store's own exact hub-or-not probe (a result longer than the bound
336
- * means MORE than the bound), never a fan-in-sized read. A constituent with
337
- * more than √N structural parents is scaffolding by §8.8's bound: " is ",
338
- * "the ". Superposing it would put a term shared by every deposit into every
339
- * profile, ALL halos would correlate, and the concept threshold's null model
340
- * (unrelated halos at 0 ± 1/√D) that §4.1's hygiene note protects would
341
- * collapse. It is still DESCENDED into — a hub chunk can contain a rare
342
- * unit — but contributes nothing itself.
335
+ * HUBS ARE THE ONE EXCLUSION, read LIMITed as `parentsFirst(n, bound+1)` — the
336
+ * store's own exact hub-or-not probe (a result longer than the bound means MORE
337
+ * than the bound), never a fan-in-sized read. A constituent with more than √N
338
+ * structural parents is scaffolding by bounded-reads.md's bound: " is ", "the
339
+ * ". Superposing it would put a term shared by every deposit into every
340
+ * profile, ALL halos would correlate, and the concept threshold's null model
341
+ * (unrelated halos at 0 ± 1/√D) that halo-sketch.md's hygiene note protects
342
+ * would collapse. It is still DESCENDED into — a hub chunk can contain a rare
343
+ * unit — but contributes nothing itself.
343
344
  *
344
345
  * Byte atoms are skipped in BOTH representations (a negative id and a stored
345
346
  * kid-less node): an atom's fan-in is the alphabet's, so it can only ever
@@ -349,32 +350,32 @@ function constituentSketch(ctx, id, k) {
349
350
  * analogy strength 0.3636 -> 0.2004, "no halo-tier company evidence",
350
351
  * test/29 C1).
351
352
  *
352
- * A FUNCTION OF THE NODE AND THE CORPUS STATE — stated precisely, because
353
- * the weaker claim is the true one. The constituents are read from the
354
- * STORE, never from the depositing tree's id map: that map holds only the
355
- * nodes THIS deposit newly interned, so a partner met a second time yielded a
356
- * profile missing exactly those constituents, the exact-partner case fell
357
- * from cosine 1 to 1/√(1+k), and the geometry stopped meaning anything.
358
- * Reading the store fixes that. It does NOT make the profile permanent: the
359
- * hub test reads fan-in against √N and both grow with training, so a partner
360
- * poured early and again late can profile differently. That residue is
361
- * confined to the hub EXCLUSION — which terms are dropped as scaffolding —
362
- * and never to which units are found, because the descent itself is now
363
- * order-independent. The drift is one-directional and benign: a term can
364
- * only ever go from contributing to being excluded as scaffolding. Replay of
365
- * a fixed training order is bit-identical, so §2.1 holds. What must not be
366
- * claimed is that a node's profile is fixed for all time; it is fixed given
367
- * the corpus that has been seen.
353
+ * A FUNCTION OF THE NODE AND THE CORPUS STATE — stated precisely, because the
354
+ * weaker claim is the true one. The constituents are read from the STORE, never
355
+ * from the depositing tree's id map: that map holds only the nodes THIS deposit
356
+ * newly interned, so a partner met a second time yielded a profile missing
357
+ * exactly those constituents, the exact-partner case fell from cosine 1 to
358
+ * 1/√(1+k), and the geometry stopped meaning anything. Reading the store fixes
359
+ * that. It does NOT make the profile permanent: the hub test reads fan-in
360
+ * against √N and both grow with training, so a partner poured early and again
361
+ * late can profile differently. That residue is confined to the hub EXCLUSION —
362
+ * which terms are dropped as scaffolding — and never to which units are found,
363
+ * because the descent itself is now order-independent. The drift is
364
+ * one-directional and benign: a term can only ever go from contributing to
365
+ * being excluded as scaffolding. Replay of a fixed training order is
366
+ * bit-identical, so determinism.md holds. What must not be claimed is that a
367
+ * node's profile is fixed for all time; it is fixed given the corpus that has
368
+ * been seen.
368
369
  *
369
- * THE NULL MODEL IS OTHERWISE UNTOUCHED (§4.1). Every term is still a seeded
370
- * function of a NODE IDENTITY, never a gist, so no byte-similarity between
371
- * partners can leak content similarity into distributional similarity. The
372
- * result is normalized, so ONE episode still pours ONE unit of mass:
373
- * {@link Store.haloMass} keeps counting episodes and every mass-based
374
- * reading is unchanged. Two partners sharing j of k discriminating
375
- * constituents meet at j/(1+k) — graded evidence, above the 1/√D noise floor
376
- * and below conceptThreshold until the overlap is most of the content, which
377
- * is the semantics "same company" should have.
370
+ * THE NULL MODEL IS OTHERWISE UNTOUCHED (halo-sketch.md). Every term is still a
371
+ * seeded function of a NODE IDENTITY, never a gist, so no byte-similarity
372
+ * between partners can leak content similarity into distributional similarity.
373
+ * The result is normalized, so ONE episode still pours ONE unit of mass: {@link
374
+ * Store.haloMass} keeps counting episodes and every mass-based reading is
375
+ * unchanged. Two partners sharing j of k discriminating constituents meet at
376
+ * j/(1+k) — graded evidence, above the 1/√D noise floor and below
377
+ * conceptThreshold until the overlap is most of the content, which is the
378
+ * semantics "same company" should have.
378
379
  *
379
380
  * Bounded: at most {@link PROFILE_VISITS} constituents are classified, each
380
381
  * by ONE LIMITed structural-parent read, so a pour costs O(1) reads in the
@@ -78,9 +78,14 @@ export interface AlignGap {
78
78
  }
79
79
  /** Extend a seed match (query offset qo ↔ candidate offset co) to its maximal
80
80
  * common run, then walk outward in both directions collecting further common
81
- * runs of at least W bytes across bounded mismatch gaps (each side ≤
82
- * chainReach). Returns the matched query spans and the mismatch pairs
83
- * between consecutive runs.
81
+ * runs of at least W bytes across mismatch gaps. Each gap's LENGTH is the
82
+ * pair's own extent (a gap cannot be longer than the bytes it spans) and the
83
+ * sweep's WORK is proportional to the bytes a run spans (the context's windows
84
+ * are indexed once, then the query's are walked) — the arity bound
85
+ * (`chainReach`) used to cap BOTH, and truncated every learned frame whose
86
+ * slot was longer. Each sweep owns its own budget, so an exhausted right
87
+ * sweep never starves the left one. Returns the matched query spans and the
88
+ * mismatch pairs between consecutive runs.
84
89
  *
85
90
  * This is the SEEDED aligner, distinct from {@link alignRuns}: that one finds
86
91
  * every run two structures share anywhere (a weave), this one reads two
@@ -31,8 +31,8 @@
31
31
  // they gate.
32
32
  import { addInto, cosine, dot, normalize, zeros } from "../vec.js";
33
33
  import { conceptThreshold, dominates, identityBar, significanceBar, } from "../geometry.js";
34
- import { bytesEqual, indexOf } from "../bytes.js";
35
- import { chainReach, leafIdRun } from "./canonical.js";
34
+ import { bytesEqual, indexOf, latin1 } from "../bytes.js";
35
+ import { leafIdRun } from "./canonical.js";
36
36
  import { foldTree, gistOf, perceive, read, resolve } from "./primitives.js";
37
37
  import { argmaxCosine, chooseAmong, chooseNext, corpusN, edgeAncestors, guidedFirst, hubBound, hubCap, sharedReachMemo, } from "./traverse.js";
38
38
  import { recognise, segment } from "./recognition.js";
@@ -237,9 +237,14 @@ export function alignGraded(ctx, query, contextBytes, querySites) {
237
237
  }
238
238
  /** Extend a seed match (query offset qo ↔ candidate offset co) to its maximal
239
239
  * common run, then walk outward in both directions collecting further common
240
- * runs of at least W bytes across bounded mismatch gaps (each side ≤
241
- * chainReach). Returns the matched query spans and the mismatch pairs
242
- * between consecutive runs.
240
+ * runs of at least W bytes across mismatch gaps. Each gap's LENGTH is the
241
+ * pair's own extent (a gap cannot be longer than the bytes it spans) and the
242
+ * sweep's WORK is proportional to the bytes a run spans (the context's windows
243
+ * are indexed once, then the query's are walked) — the arity bound
244
+ * (`chainReach`) used to cap BOTH, and truncated every learned frame whose
245
+ * slot was longer. Each sweep owns its own budget, so an exhausted right
246
+ * sweep never starves the left one. Returns the matched query spans and the
247
+ * mismatch pairs between consecutive runs.
243
248
  *
244
249
  * This is the SEEDED aligner, distinct from {@link alignRuns}: that one finds
245
250
  * every run two structures share anywhere (a weave), this one reads two
@@ -253,7 +258,21 @@ export function alignGraded(ctx, query, contextBytes, querySites) {
253
258
  * window test); {@link frameSlots} takes the other reading. */
254
259
  export function alignAround(ctx, q, c, qo, co) {
255
260
  const W = ctx.space.maxGroup;
256
- const reachCap = chainReach(W);
261
+ // THE GAP LENGTH IS THE PAIR'S OWN EXTENT; THE WORK IS BUDGETED.
262
+ //
263
+ // The sweep walks (queryGap, contextGap) pairs by ASCENDING total, so reaching
264
+ // a gap of size G costs about G²/2 pairs. Bounding the LENGTH by the write
265
+ // side's arity (`chainReach(W)` = 16) therefore truncated every learned frame
266
+ // whose slot is longer — measured: `bindReference` reported the cap at 18, 24,
267
+ // 30 and 36 bytes and `recall` answered with ANOTHER instance's filler — while
268
+ // removing the bound outright took the corpus-cost guard (test/89) from
269
+ // milliseconds to 68 seconds.
270
+ //
271
+ // Bounding the PAIRS keeps a call's cost constant however long the pair is,
272
+ // and the ascending order means an exhausted budget drops the FAR
273
+ // continuations and never the near ones — the same degradation recognition.ts
274
+ // documents for its canon budget. Length and work are different questions;
275
+ // this is the one place they were conflated.
257
276
  // Maximal run around the seed.
258
277
  let qs = qo, ss = co;
259
278
  while (qs > 0 && ss > 0 && q[qs - 1] === c[ss - 1]) {
@@ -267,8 +286,66 @@ export function alignAround(ctx, q, c, qo, co) {
267
286
  }
268
287
  const matched = [[qs, qe]];
269
288
  const gaps = [];
270
- // The next common run of ≥ W bytes past (qi, si), with each side's gap
271
- // bounded by chainReach; smallest total gap wins (nearest continuation).
289
+ // THE SWEEP IS STRUCTURAL, NOT ENUMERATIVE.
290
+ //
291
+ // The criterion is unchanged: the next common run, MINIMUM TOTAL GAP, ties to
292
+ // the smaller query gap. What changed is how it is found. Enumerating
293
+ // (queryGap, contextGap) pairs by ascending total reaches a run at total t in
294
+ // about t²/2 pairs — and that quadratic shape, not the reach, was the cost
295
+ // problem: capping the pairs dropped reach (a legitimate 24-byte slot stopped
296
+ // being found), while leaving them uncapped cost 68 seconds on the corpus
297
+ // guard. Neither is the answer, because the answer is the algorithm.
298
+ //
299
+ // The context's windows are indexed ONCE, for lengths 1..W — W being the
300
+ // geometry's own unit of composition, so nothing is chosen here. Each step
301
+ // then walks the query's windows outward from the anchor: for a given query
302
+ // gap the nearest context gap that continues a run is one O(1) lookup, and the
303
+ // walk stops the moment the query gap alone exceeds the best total already
304
+ // found. So the work is proportional to the bytes the run SPANS. No budget,
305
+ // no cap, no number: a long slot is reached, and its price is already the
306
+ // ladder's (its bytes are unaccounted, so the search pays PASS per byte).
307
+ const index = [];
308
+ for (let len = 1; len <= W; len++) {
309
+ const m = new Map();
310
+ for (let o = 0; o + len <= c.length; o++) {
311
+ const key = latin1(c.subarray(o, o + len));
312
+ const at = m.get(key);
313
+ if (at === undefined)
314
+ m.set(key, [o]);
315
+ else
316
+ at.push(o);
317
+ }
318
+ index.push(m);
319
+ }
320
+ /** Smallest listed offset at or after `from`, or -1. */
321
+ const fromAt = (list, from) => {
322
+ let lo = 0, hi = list.length - 1, best = -1;
323
+ while (lo <= hi) {
324
+ const mid = (lo + hi) >> 1;
325
+ if (list[mid] >= from) {
326
+ best = list[mid];
327
+ hi = mid - 1;
328
+ }
329
+ else
330
+ lo = mid + 1;
331
+ }
332
+ return best;
333
+ };
334
+ /** Largest listed offset at or before `to`, or -1. */
335
+ const toAt = (list, to) => {
336
+ let lo = 0, hi = list.length - 1, best = -1;
337
+ while (lo <= hi) {
338
+ const mid = (lo + hi) >> 1;
339
+ if (list[mid] <= to) {
340
+ best = list[mid];
341
+ lo = mid + 1;
342
+ }
343
+ else
344
+ hi = mid - 1;
345
+ }
346
+ return best;
347
+ };
348
+ /** Length of the common run STARTING at (qi, si). */
272
349
  const runLenAt = (qi, si) => {
273
350
  let n = 0;
274
351
  while (qi + n < q.length && si + n < c.length && q[qi + n] === c[si + n]) {
@@ -276,69 +353,76 @@ export function alignAround(ctx, q, c, qo, co) {
276
353
  }
277
354
  return n;
278
355
  };
279
- // RIGHT sweep.
280
- let qi = qe, si = se;
281
- for (;;) {
282
- let found = false;
283
- for (let total = 1; total <= 2 * reachCap && !found; total++) {
284
- for (let gq = 0; gq <= Math.min(total, reachCap); gq++) {
285
- const gs = total - gq;
286
- if (gs > reachCap)
356
+ /** Length of the common run ENDING at (qi, si). */
357
+ const runLenBefore = (qi, si) => {
358
+ let n = 0;
359
+ while (n < qi && n < si && q[qi - 1 - n] === c[si - 1 - n])
360
+ n++;
361
+ return n;
362
+ };
363
+ /** The next run outward from an anchor, or null when the bytes run out. */
364
+ const nextRun = (qi, si, forward) => {
365
+ const qLim = forward ? q.length - qi : qi;
366
+ let best = null;
367
+ for (let gq = 0; gq < qLim; gq++) {
368
+ // No later query gap can beat a total already found.
369
+ if (best !== null && gq > best.gq + best.gs)
370
+ break;
371
+ const left = qLim - gq;
372
+ // A run of >= W bytes, or — when the query itself ends inside one window —
373
+ // the run that REACHES that end. Exactly the acceptance the sweep had.
374
+ const lens = left >= W ? [W] : [left];
375
+ for (const len of lens) {
376
+ const key = latin1(q.subarray(forward ? qi + gq : qi - gq - len, forward ? qi + gq + len : qi - gq));
377
+ const list = index[len - 1].get(key);
378
+ if (list === undefined)
287
379
  continue;
288
- if (qi + gq >= q.length || si + gs >= c.length)
380
+ const o = forward ? fromAt(list, si) : toAt(list, si - len);
381
+ if (o < 0)
289
382
  continue;
290
- const n = runLenAt(qi + gq, si + gs);
291
- if (n >= W || qi + gq + n === q.length) {
292
- if (n === 0)
293
- continue;
294
- if (gq > 0 || gs > 0) {
295
- gaps.push({ qs: qi, qe: qi + gq, cs: si, ce: si + gs });
383
+ const n = forward
384
+ ? runLenAt(qi + gq, o)
385
+ : runLenBefore(qi - gq, o + len);
386
+ if (n < 1)
387
+ continue;
388
+ if (forward ? n >= W || qi + gq + n === q.length : n >= W || n === qi - gq) {
389
+ const gs = forward ? o - si : si - len - o;
390
+ if (best === null || gq + gs < best.gq + best.gs) {
391
+ best = { gq, gs, n };
296
392
  }
297
- matched.push([qi + gq, qi + gq + n]);
298
- qi = qi + gq + n;
299
- si = si + gs + n;
300
- found = true;
301
393
  break;
302
394
  }
303
395
  }
304
396
  }
305
- if (!found)
397
+ return best;
398
+ };
399
+ // RIGHT sweep.
400
+ let qi = qe, si = se;
401
+ for (;;) {
402
+ const step = nextRun(qi, si, true);
403
+ if (step === null)
306
404
  break;
405
+ if (step.gq > 0 || step.gs > 0) {
406
+ gaps.push({ qs: qi, qe: qi + step.gq, cs: si, ce: si + step.gs });
407
+ }
408
+ matched.push([qi + step.gq, qi + step.gq + step.n]);
409
+ qi = qi + step.gq + step.n;
410
+ si = si + step.gs + step.n;
307
411
  }
308
- // LEFT sweep (mirror).
412
+ // LEFT sweep (mirror): an independent walk, so an exhausted right side can
413
+ // never starve it (pinned by test/114).
309
414
  qi = qs;
310
415
  si = ss;
311
416
  for (;;) {
312
- let found = false;
313
- for (let total = 1; total <= 2 * reachCap && !found; total++) {
314
- for (let gq = 0; gq <= Math.min(total, reachCap); gq++) {
315
- const gs = total - gq;
316
- if (gs > reachCap)
317
- continue;
318
- if (qi - gq <= 0 || si - gs <= 0)
319
- continue;
320
- // Run ENDING at (qi - gq, si - gs).
321
- let n = 0;
322
- while (n < qi - gq && n < si - gs &&
323
- q[qi - gq - 1 - n] === c[si - gs - 1 - n]) {
324
- n++;
325
- }
326
- if (n >= W || n === qi - gq) {
327
- if (n === 0)
328
- continue;
329
- if (gq > 0 || gs > 0) {
330
- gaps.push({ qs: qi - gq, qe: qi, cs: si - gs, ce: si });
331
- }
332
- matched.push([qi - gq - n, qi - gq]);
333
- qi = qi - gq - n;
334
- si = si - gs - n;
335
- found = true;
336
- break;
337
- }
338
- }
339
- }
340
- if (!found)
417
+ const step = nextRun(qi, si, false);
418
+ if (step === null)
341
419
  break;
420
+ if (step.gq > 0 || step.gs > 0) {
421
+ gaps.push({ qs: qi - step.gq, qe: qi, cs: si - step.gs, ce: si });
422
+ }
423
+ matched.push([qi - step.gq - step.n, qi - step.gq]);
424
+ qi = qi - step.gq - step.n;
425
+ si = si - step.gs - step.n;
342
426
  }
343
427
  return { matched, gaps };
344
428
  }
@@ -934,24 +1018,25 @@ export async function project(ctx, id, guide) {
934
1018
  }
935
1019
  // ── The span-shape family ───────────────────────────────────────────────────
936
1020
  //
937
- // "Is this answer drawn from this context?" has TWO formally distinct
938
- // readings, and the pair plus the anchor classifier built on them are SHARED
939
- // machinery — extraction proposes span-shaped exemplars with them, the
940
- // shared `Precomputed.spanShapedOf` container computes them, and fusion
941
- // (reasoning.ts) gates on the strict one. They lived inside
942
- // mechanisms/extraction.ts, so `pipeline-mechanism.ts` and `reasoning.ts`
943
- // both had to import back OUT of a specific mechanism — an inversion the
944
- // mechanism market forbids (AGENTS §2.6: the shared contract may not depend
945
- // on any one mechanism; §2.5: a shared matcher belongs to this family, never
946
- // to a mechanism's private helpers). Deleting extraction must not break the
947
- // shared container, so they live here.
1021
+ // "Is this answer drawn from this context?" has TWO formally distinct readings,
1022
+ // and the pair plus the anchor classifier built on them are SHARED machinery —
1023
+ // extraction proposes span-shaped exemplars with them, the shared
1024
+ // `Precomputed.spanShapedOf` container computes them, and fusion (reasoning.ts)
1025
+ // gates on the strict one. They lived inside mechanisms/extraction.ts, so
1026
+ // `pipeline-mechanism.ts` and `reasoning.ts` both had to import back OUT of a
1027
+ // specific mechanism — an inversion the mechanism market forbids: the shared
1028
+ // contract may not depend on any one mechanism (mechanism-market.md), and a
1029
+ // shared matcher belongs to this family (match-project.md), never to a
1030
+ // mechanism's private helpers. Deleting extraction must not break the shared
1031
+ // container, so they live here.
948
1032
  //
949
1033
  // • isSpanShaped — the OPEN reading (sparse in-order embedding).
950
1034
  // • containsSpan — the STRICT reading (contiguous run or resolved node).
951
1035
  // • skillExemplar — classify one anchor into (context, answer) using them.
952
1036
  //
953
- // The two readings are NOT interchangeable; AGENTS §2.5 pins the distinction
954
- // and each function's own doc states what breaks if it is substituted.
1037
+ // The two readings are NOT interchangeable; match-project.md pins the
1038
+ // distinction and each function's own doc states what breaks if it is
1039
+ // substituted.
955
1040
  /** Check whether an anchor is a span-shaped skill exemplar: it represents a
956
1041
  * fact whose context and answer together form a span-in-context pattern.
957
1042
  * If the anchor has a nextOf continuation, that is the answer and the anchor
@@ -16,7 +16,7 @@ import { analogyStrength, follow, project, reverseContext, sharedFrameStrengthOf
16
16
  import { joinWithBridge } from "../resonance.js";
17
17
  import { restatesQuery } from "../reasoning.js";
18
18
  import { CONCEPT, STEP } from "../graph-search.js";
19
- import { concat2, indexOf } from "../../bytes.js";
19
+ import { indexOf } from "../../bytes.js";
20
20
  import { consensusFloor, dominates } from "../../geometry.js";
21
21
  import { unexplainedLabel, unexplainedSpans, } from "../rationale.js";
22
22
  import { rItem, rNode } from "../trace.js";
@@ -523,7 +523,23 @@ export async function counterfactualTransfer(ctx, query, pre) {
523
523
  const fwd = await follow(ctx, proj.anchor, qv);
524
524
  if (fwd !== null && indexOf(answer, fwd, 0) < 0 &&
525
525
  !restatesQuery(query, fwd)) {
526
- answer = concat2(answer, fwd);
526
+ // THROUGH THE SHARED JOINER, not a bare concatenation.
527
+ //
528
+ // `joinWithBridge` is the composition step every out-of-search assembly
529
+ // shares (multi-topic fusion, CAST's substitution and comparison): it
530
+ // asks the corpus for a learnt connector between the pieces and, on a
531
+ // miss, joins them BARE **and says so** — the `bridgeMiss` step (see
532
+ // resonance.ts). This site bypassed it, and that is the whole of the
533
+ // gluing the study measured: `"Steel is hard"` + `"wet"` came back as
534
+ // `"hardwet"`, `"eva director father"` + `"The father of…"` as
535
+ // `"fatherThe"` — compositions no rationale could show, because the one
536
+ // step that made them left no trace.
537
+ //
538
+ // Routing it through the shared joiner is the instrumentation fix that
539
+ // comes first: a bare join stays possible (the house rule is "joined
540
+ // bare, never silent") but it is now VISIBLE, and an attested connector
541
+ // is used when the corpus has one.
542
+ answer = await joinWithBridge(ctx, answer, fwd);
527
543
  }
528
544
  ctx.trace?.step("projectCounterfactual", [
529
545
  rItem(filler, "filler", subj.point.anchor),
@@ -4,7 +4,7 @@
4
4
  // Cover consumes recognition directly (its axioms are the query's own
5
5
  // decomposition) plus the computed spans any parse()-bearing mechanism
6
6
  // contributed: computed spans MASK colliding recognised sites and enter the
7
- // search at zero cost ("computation always wins", §16.3) — which is also why
7
+ // search at zero cost ("computation always wins", alu.md) — which is also why
8
8
  // cover runs FIRST in defaultMechanisms: a computed-backed cover becomes a
9
9
  // near-zero-cost incumbent that prunes the other mechanisms through the
10
10
  // ordinary admissible-floor check, with no extension special-case anywhere.
@@ -59,22 +59,23 @@ export async function resolveConnectors(ctx, sites, query) {
59
59
  return true;
60
60
  const continuations = ctx.store.nextFirst(s.payload, hubBound(ctx));
61
61
  return !continuations.some((answer) => {
62
- // PREFIX-CAPPED (AGENTS §2.8): a candidate longer than the query cannot
63
- // occur INSIDE it, so read one byte past the query's length — enough to
64
- // detect the overflow — and reject without reconstructing the rest.
65
- // The `+ 1` is what makes the test exact rather than a truncation: a
66
- // result of exactly `query.length + 1` bytes is known to be too long,
67
- // and anything shorter is the candidate's COMPLETE content, so the
68
- // substring test below is the same test as before. (The same overflow
69
- // probe bridge.ts:256 already uses.)
62
+ // PREFIX-CAPPED (bounded-reads.md): a candidate longer than the query
63
+ // cannot occur INSIDE it, so read one byte past the query's length —
64
+ // enough to detect the overflow — and reject without reconstructing the
65
+ // rest. The `+ 1` is what makes the test exact rather than a
66
+ // truncation: a result of exactly `query.length + 1` bytes is known to
67
+ // be too long, and anything shorter is the candidate's COMPLETE
68
+ // content, so the substring test below is the same test as before. (The
69
+ // same overflow probe bridge.ts:256 already uses.)
70
70
  //
71
71
  // This loop runs up to hubBound(ctx) = √N reads PER SITE, and only on a
72
72
  // multi-turn response — `answeredSpans` is empty for a plain respond(),
73
- // so the probe does not execute there. The cap cannot reduce the read
73
+ // so the probe does not execute there. The cap cannot reduce the read
74
74
  // COUNT — only a semantic change to the "already answered" test could —
75
75
  // but it bounds each read by the query instead of by the corpus, which
76
- // is what §2.8 asks for and what rescues a SHORT query: at 3 bytes this
77
- // reads 4 bytes per candidate instead of the ~231 it averaged before.
76
+ // is what bounded-reads.md asks for and what rescues a SHORT query: at
77
+ // 3 bytes this reads 4 bytes per candidate instead of the ~231 it
78
+ // averaged before.
78
79
  const bytes = read(ctx, answer, query.length + 1);
79
80
  return bytes.length <= query.length && indexOf(query, bytes, 0) >= 0;
80
81
  });
@@ -82,6 +83,8 @@ export async function resolveConnectors(ctx, sites, query) {
82
83
  const bridgePair = async (l, r) => {
83
84
  if (l === r || links.has(l + "," + r))
84
85
  return;
86
+ if (ctx.meter)
87
+ ctx.meter.coverBridges++;
85
88
  const link = await bridge(ctx, read(ctx, l), read(ctx, r));
86
89
  if (link !== null)
87
90
  links.set(l + "," + r, link);
@@ -121,6 +124,10 @@ export async function resolveConnectors(ctx, sites, query) {
121
124
  // plus one W-quantum of glue per joint — pass that allowance so the
122
125
  // bridge's phrase-scale cap admits the whole learnt run.
123
126
  const allowance = middleBytes + (m + 1) * W;
127
+ if (ctx.meter) {
128
+ ctx.meter.coverBridges++;
129
+ ctx.meter.coverAllowanceBytes += allowance;
130
+ }
124
131
  const interior = await bridge(ctx, first.bytes, orderedNodes[m].bytes, allowance);
125
132
  if (interior !== null)
126
133
  links.set(key, interior);
@@ -1,15 +1,15 @@
1
1
  // mechanisms/prefix-completion.ts — Grounding a query that IS the opening of a
2
2
  // trained form (Grounding V).
3
3
  //
4
- // A MECHANISM, NOT A TIER. This used to run inside recall's refusal path, in
5
- // a fixed if-chain that first-match-wins — the shape CAST was refactored away
6
- // from, where placement rather than the cost ladder decided. Its claim is
4
+ // A MECHANISM, NOT A TIER. This used to run inside recall's refusal path, in a
5
+ // fixed if-chain that first-match-wins — the shape CAST was refactored away
6
+ // from, where placement rather than the cost ladder decided. Its claim is
7
7
  // maximal (every query byte literally matched, from offset zero, against a
8
8
  // trained form) at one STEP, so as a market candidate it competes honestly and
9
- // the decider weighs it like everything else. It is registered LAST: recall's
9
+ // the decider weighs it like everything else. It is registered LAST: recall's
10
10
  // exact self-match makes an IDENTITY claim about the query while this makes a
11
- // CONTAINMENT one, and on an exact grade tie the identity claim is the
12
- // stronger evidence — the same ordering §2.3's ladders use.
11
+ // CONTAINMENT one, and on an exact grade tie the identity claim is the stronger
12
+ // evidence — the same ordering exact-vs-approximate.md's ladders use.
13
13
  //
14
14
  // Its SUPPLY moved too, and further: `formsOpenedBy` (traverse.ts) answers a
15
15
  // question about the STORE — "which trained forms does this byte run open?" —
@@ -49,12 +49,12 @@
49
49
  //
50
50
  // So this is a RETRIEVABILITY gap, not a semantic one, and the ANN is the wrong
51
51
  // instrument for it: a proper prefix's gist cannot rank its own continuation.
52
- // The repair is CONTENT-ADDRESSED (§2.3) — `formsOpenedBy` (traverse.ts) reads
53
- // the leaf-id WINDOW index the write side already maintains and answers "which
54
- // trained forms does this byte run open?" in a bounded √N walk. That is this
55
- // mechanism's first supply. The response's memoised top-k `resonance()` is the
56
- // second, for prefixes long enough that the gist still ranks the form; it is
57
- // read, never re-issued.
52
+ // The repair is CONTENT-ADDRESSED (exact-vs-approximate.md) — `formsOpenedBy`
53
+ // (traverse.ts) reads the leaf-id WINDOW index the write side already maintains
54
+ // and answers "which trained forms does this byte run open?" in a bounded √N
55
+ // walk. That is this mechanism's first supply. The response's memoised top-k
56
+ // `resonance()` is the second, for prefixes long enough that the gist still
57
+ // ranks the form; it is read, never re-issued.
58
58
  //
59
59
  // AN EXHAUSTIVE ANN LIST IS NOT A SUPPLY HERE, AND WAS REMOVED. This tier once
60
60
  // read `Precomputed.wideResonance()` — a full-index `resonate(guide, √N,
@@ -227,20 +227,20 @@ export const prefixMechanism = {
227
227
  return STEP;
228
228
  },
229
229
  async run(ctx, query, pre) {
230
- // ONE SUPPLY PASS, not a two-tier `??`. The window index (exact,
230
+ // ONE SUPPLY PASS, not a two-tier `??`. The window index (exact,
231
231
  // content-addressed) and the response's memoised top-k (approximate) are
232
- // concatenated and the three guards decide ONCE over the union. A
232
+ // concatenated and the three guards decide ONCE over the union. A
233
233
  // first-then-fallback chain would let the APPROXIMATE tier override the
234
- // EXACT one (§2.3): when formsOpenedBy finds two continuations, guard 3
235
- // returns null and the fallback re-runs the guards on resonance's top-k
236
- // alone — which, seeing only one of the two forms, would voice it. That is
237
- // precisely the disagreement-suppression guard 3 exists to prevent, and it
238
- // is the exact tier's ambiguity being washed away by the approximate tier.
239
- // Evaluating the union means a disagreement the window index saw can never
240
- // be hidden by what the ANN happens to rank. The ANN read is the
241
- // response's ONE memoised top-k (§2.11), already paid by recall's refusal
242
- // path on the queries where this mechanism fires, so reading it here is not
243
- // a second index scan.
234
+ // EXACT one (exact-vs-approximate.md): when formsOpenedBy finds two
235
+ // continuations, guard 3 returns null and the fallback re-runs the guards
236
+ // on resonance's top-k alone — which, seeing only one of the two forms,
237
+ // would voice it. That is precisely the disagreement-suppression guard 3
238
+ // exists to prevent, and it is the exact tier's ambiguity being washed away
239
+ // by the approximate tier. Evaluating the union means a disagreement the
240
+ // window index saw can never be hidden by what the ANN happens to rank. The
241
+ // ANN read is the response's ONE memoised top-k (memoization.md), already
242
+ // paid by recall's refusal path on the queries where this mechanism fires,
243
+ // so reading it here is not a second index scan.
244
244
  const ids = [
245
245
  ...formsOpenedBy(ctx, query),
246
246
  ...(await pre.resonance()).map((h) => h.id),