@hviana/sema 0.8.0 → 0.8.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (119) hide show
  1. package/AGENTS.md +22 -1
  2. package/DATASETS.md +1 -1
  3. package/dist/example/train_base/config.js +2 -2
  4. package/dist/example/train_base/corpora/massive.js +1 -1
  5. package/dist/example/train_base/readers.js +1 -1
  6. package/dist/src/config.d.ts +17 -0
  7. package/dist/src/config.js +18 -0
  8. package/dist/src/geometry.d.ts +10 -10
  9. package/dist/src/geometry.js +25 -24
  10. package/dist/src/meter.d.ts +29 -12
  11. package/dist/src/meter.js +58 -14
  12. package/dist/src/mind/attention.js +12 -12
  13. package/dist/src/mind/bridge.d.ts +8 -8
  14. package/dist/src/mind/bridge.js +33 -32
  15. package/dist/src/mind/corpus.d.ts +40 -0
  16. package/dist/src/mind/corpus.js +149 -0
  17. package/dist/src/mind/graph-search.d.ts +7 -8
  18. package/dist/src/mind/graph-search.js +244 -32
  19. package/dist/src/mind/index.d.ts +3 -1
  20. package/dist/src/mind/index.js +1 -0
  21. package/dist/src/mind/junction.d.ts +1 -1
  22. package/dist/src/mind/junction.js +8 -8
  23. package/dist/src/mind/learning.js +36 -35
  24. package/dist/src/mind/match.d.ts +8 -3
  25. package/dist/src/mind/match.js +156 -71
  26. package/dist/src/mind/mechanisms/cast.js +18 -2
  27. package/dist/src/mind/mechanisms/cover.js +19 -12
  28. package/dist/src/mind/mechanisms/prefix-completion.js +24 -24
  29. package/dist/src/mind/mechanisms/recall.js +38 -40
  30. package/dist/src/mind/mechanisms/reference.js +16 -16
  31. package/dist/src/mind/mind.d.ts +61 -7
  32. package/dist/src/mind/mind.js +72 -2
  33. package/dist/src/mind/pipeline-mechanism.d.ts +10 -8
  34. package/dist/src/mind/pipeline-mechanism.js +25 -21
  35. package/dist/src/mind/pipeline.d.ts +9 -9
  36. package/dist/src/mind/pipeline.js +49 -29
  37. package/dist/src/mind/primitives.d.ts +5 -5
  38. package/dist/src/mind/primitives.js +5 -5
  39. package/dist/src/mind/reasoning.d.ts +5 -1
  40. package/dist/src/mind/reasoning.js +54 -1
  41. package/dist/src/mind/recognition.d.ts +14 -13
  42. package/dist/src/mind/recognition.js +23 -23
  43. package/dist/src/mind/resonance.js +21 -21
  44. package/dist/src/mind/traverse.d.ts +54 -52
  45. package/dist/src/mind/traverse.js +83 -73
  46. package/dist/src/mind/types.d.ts +26 -4
  47. package/dist/src/store.d.ts +12 -12
  48. package/dist/src/store.js +12 -12
  49. package/docs/INDEX.md +2 -2
  50. package/docs/architecture/exact-vs-approximate.md +2 -1
  51. package/docs/architecture/fold-contract.md +1 -1
  52. package/docs/failures/tempting-but-wrong.md +33 -5
  53. package/docs/harness/gates.md +7 -7
  54. package/example/train_base/config.ts +2 -2
  55. package/example/train_base/corpora/massive.ts +1 -1
  56. package/example/train_base/readers.ts +1 -1
  57. package/jsr.json +1 -1
  58. package/package.json +1 -1
  59. package/src/config.ts +35 -0
  60. package/src/geometry.ts +25 -24
  61. package/src/meter.ts +61 -14
  62. package/src/mind/attention.ts +12 -12
  63. package/src/mind/bridge.ts +33 -32
  64. package/src/mind/corpus.ts +202 -0
  65. package/src/mind/graph-search.ts +261 -31
  66. package/src/mind/index.ts +8 -1
  67. package/src/mind/junction.ts +8 -8
  68. package/src/mind/learning.ts +36 -35
  69. package/src/mind/match.ts +163 -73
  70. package/src/mind/mechanisms/cast.ts +17 -1
  71. package/src/mind/mechanisms/cover.ts +18 -12
  72. package/src/mind/mechanisms/prefix-completion.ts +24 -24
  73. package/src/mind/mechanisms/recall.ts +38 -40
  74. package/src/mind/mechanisms/reference.ts +16 -16
  75. package/src/mind/mind.ts +129 -7
  76. package/src/mind/pipeline-mechanism.ts +25 -21
  77. package/src/mind/pipeline.ts +63 -38
  78. package/src/mind/primitives.ts +5 -5
  79. package/src/mind/reasoning.ts +55 -0
  80. package/src/mind/recognition.ts +23 -23
  81. package/src/mind/resonance.ts +21 -21
  82. package/src/mind/traverse.ts +83 -73
  83. package/src/mind/types.ts +30 -4
  84. package/src/store.ts +20 -20
  85. package/test/08-storage.test.mjs +1 -1
  86. package/test/100-complete-grounding-trace.test.mjs +109 -0
  87. package/test/101-alignment-gap-bound.test.mjs +106 -0
  88. package/test/102-production-composes-at-scale.test.mjs +110 -0
  89. package/test/103-alignment-gap-budget.test.mjs +89 -0
  90. package/test/104-composition-is-reported.test.mjs +90 -0
  91. package/test/105-derive-through-reports-its-refusal.test.mjs +113 -0
  92. package/test/106-the-join-fires.test.mjs +94 -0
  93. package/test/107-the-join-is-counted.test.mjs +81 -0
  94. package/test/108-the-join-chains.test.mjs +78 -0
  95. package/test/109-the-pivot-is-counted.test.mjs +60 -0
  96. package/test/110-the-reasoner-stops-when-the-question-is-answered.test.mjs +91 -0
  97. package/test/111-the-cover-assembly-is-counted.test.mjs +74 -0
  98. package/test/112-the-exploration-does-not-grow-with-the-hub.test.mjs +89 -0
  99. package/test/113-the-rationale-payload-is-bounded.test.mjs +84 -0
  100. package/test/114-alignment-budget-is-per-sweep.test.mjs +93 -0
  101. package/test/116-the-extension-is-gated-by-the-pipelines-own-remainder.test.mjs +100 -0
  102. package/test/117-corpus-search.test.mjs +171 -0
  103. package/test/14-scaling.test.mjs +10 -7
  104. package/test/35-prefix-edge.test.mjs +1 -1
  105. package/test/40-choosenext-scale-guard.test.mjs +16 -17
  106. package/test/56-bridge-identity-admission.test.mjs +6 -6
  107. package/test/70-prefix-completion.test.mjs +4 -3
  108. package/test/72-prefix-candidate-supply.test.mjs +3 -3
  109. package/test/73-scaffolding-only-bridge-abstains.test.mjs +6 -6
  110. package/test/75-multiturn-context-optimisation.test.mjs +5 -5
  111. package/test/76-reference-binding.test.mjs +6 -1
  112. package/test/84-composed-answer-honesty.test.mjs +5 -6
  113. package/test/88-dependency-footprint.test.mjs +1 -1
  114. package/test/89-completion-recursion.test.mjs +47 -19
  115. package/test/90-connector-read-cap.test.mjs +10 -8
  116. package/test/93-regime-prediction.test.mjs +10 -10
  117. package/test/94-cross-region-budget.test.mjs +2 -2
  118. package/test/95-wide-resonance-removed.test.mjs +8 -7
  119. package/test/96-bytes-walk-termination.test.mjs +3 -3
@@ -0,0 +1,149 @@
1
+ // corpus.ts — read the trained memory back out of the DAG, AS DATA.
2
+ //
3
+ // A trained experience pair IS one continuation edge: `src` is the context that
4
+ // was deposited, `dst` is what the mind learnt follows it. Reading them back
5
+ // uses the store's own structure and its own indexes — no auxiliary index is
6
+ // built, nothing is written, and NOTHING HERE KNOWS ABOUT TEXT: this layer takes
7
+ // bytes and returns bytes. The text case is one helper on the Mind
8
+ // (`searchCorpusText`), which encodes, calls this, and decodes.
9
+ //
10
+ // WHERE EACH STAGE COMES FROM (ported from the demo's `explore.ts`, which
11
+ // hand-rolled its own resolution):
12
+ //
13
+ // 1. PERCEIVE, CONTENT-ADDRESS and ADMIT the query — `recognise()`, the SAME
14
+ // machinery an answer goes through. It returns the sites: the query spans
15
+ // that content-addressed to a stored node that can lead somewhere. The
16
+ // demo's own recursive `findLeaf`/`findBranch` walk was a second
17
+ // implementation of exactly this, and it is NOT ported.
18
+ // 2. CLIMB the structural `kid` table from each resolved site to the
19
+ // edge-bearing contexts above it (`edgeAncestors`, traverse.ts), weighting
20
+ // each context by how much query content reached it.
21
+ // 3. READ the continuation off the edge table (`nextFirst`).
22
+ //
23
+ // COST is set by how much of the QUERY resolves, never by the size of the
24
+ // store: the sites are what recognition already found, the climb is bounded by
25
+ // the declared `corpusClimbs`/`corpusContextsPerClimb`, and every store call is
26
+ // a point probe or a capped read. All work is accounted by the store's own
27
+ // meter hooks when a response's meter is open — there is no second instrument.
28
+ //
29
+ // WHAT THIS IS NOT. Exact content addressing, not fuzzy keyword search: a query
30
+ // shares results with a stored note when it shares actual chunk-aligned content
31
+ // with it. An arbitrary mid-word fragment resolves to nothing, and the honest
32
+ // answer there is "nothing matched" — which is why `sampleCorpus` exists, and
33
+ // why the miss is reported as a STATE rather than as prose (the text helper
34
+ // turns it into words).
35
+ import { recognise } from "./recognition.js";
36
+ import { edgeAncestors } from "./traverse.js";
37
+ /** A context node as a pair, or null when it carries no continuation. */
38
+ function pairOf(ctx, id, matchedBytes) {
39
+ const outs = ctx.store.nextFirst(id, 1);
40
+ if (outs.length === 0)
41
+ return null;
42
+ const cap = ctx.cfg.corpusPreviewBytes;
43
+ const context = ctx.store.bytesPrefix(id, cap + 1);
44
+ const continuation = ctx.store.bytesPrefix(outs[0], cap + 1);
45
+ const contextTruncated = context.length > cap;
46
+ const continuationTruncated = continuation.length > cap;
47
+ return {
48
+ context: contextTruncated ? context.subarray(0, cap) : context,
49
+ continuation: continuationTruncated
50
+ ? continuation.subarray(0, cap)
51
+ : continuation,
52
+ contextId: id,
53
+ continuationId: outs[0],
54
+ matchedBytes,
55
+ contextTruncated,
56
+ continuationTruncated,
57
+ };
58
+ }
59
+ /** Which stored notes does this query reach? BYTES in, BYTES out.
60
+ *
61
+ * Exact content addressing through the machinery that already exists: the
62
+ * query's recognised sites are the resolved subtrees, the climb goes up from
63
+ * the biggest first, and a pair is a context that carries a continuation. */
64
+ export function searchCorpus(ctx, queryBytes, limit) {
65
+ const store = ctx.store;
66
+ const want = Math.max(1, Math.min(limit ?? ctx.cfg.corpusContextsPerClimb, ctx.cfg.corpusLimitMax));
67
+ // A resolved subtree must account for at least one window (W): a single
68
+ // character resolves against almost any store and means nothing. W is the
69
+ // mind's own line between chance and evidence — derived, never declared.
70
+ const floor = ctx.space.maxGroup;
71
+ const resolved = recognise(ctx, queryBytes).sites
72
+ .map((s) => ({ id: s.payload, len: store.contentLen(s.payload, 512) }))
73
+ .filter((r) => r.len >= floor);
74
+ // Biggest first, lowest id breaking ties: a clause is evidence, a character is
75
+ // noise, and equal evidence must not be decided by iteration order.
76
+ const byLength = resolved
77
+ .sort((a, b) => b.len - a.len || a.id - b.id)
78
+ .slice(0, ctx.cfg.corpusClimbs);
79
+ // Weight each context by how much query content reached it.
80
+ const weight = new Map();
81
+ for (const { id, len } of byLength) {
82
+ for (const root of edgeAncestors(ctx, id, ctx.cfg.corpusContextsPerClimb).roots) {
83
+ weight.set(root, (weight.get(root) ?? 0) + len);
84
+ }
85
+ }
86
+ const pairs = [];
87
+ const ranked = [...weight.entries()].sort((a, b) => b[1] - a[1] || a[0] - b[0]);
88
+ for (const [id, w] of ranked) {
89
+ if (pairs.length >= want)
90
+ break;
91
+ const pair = pairOf(ctx, id, w);
92
+ if (pair)
93
+ pairs.push(pair);
94
+ }
95
+ return {
96
+ pairs,
97
+ resolved: byLength.length,
98
+ reached: weight.size,
99
+ totalContexts: store.edgeSourceCount(),
100
+ browsed: false,
101
+ miss: pairs.length > 0
102
+ ? "matched"
103
+ : weight.size === 0
104
+ ? "nothing-resolved"
105
+ : "no-continuations",
106
+ };
107
+ }
108
+ /** Browse real pairs, striding the id space so the sample is spread rather than
109
+ * one local cluster. DETERMINISTIC: `from` is the caller's own offset, so
110
+ * browsing twice with different offsets shows different notes without a random
111
+ * draw (the demo drew `Math.random()`, which the engine cannot do — same seed,
112
+ * same order, same query must mean the same answer). */
113
+ export function sampleCorpus(ctx, limit, from = 0) {
114
+ const store = ctx.store;
115
+ const want = Math.max(1, Math.min(limit ?? ctx.cfg.corpusContextsPerClimb, ctx.cfg.corpusLimitMax));
116
+ const total = store.nodeCount();
117
+ const probes = ctx.cfg.corpusSampleProbes;
118
+ const floorBytes = ctx.cfg.corpusSampleFloorBytes;
119
+ const pairs = [];
120
+ // EACH CONTEXT AT MOST ONCE. Striding the id space revisits ids when the
121
+ // store is small relative to the probe budget (measured: a 160-node store
122
+ // returned the SAME pair six times for `limit: 6`), and a browse that repeats
123
+ // itself is not a browse. The demo had the same hole; it is invisible only on
124
+ // a store far larger than the probe budget.
125
+ const seen = new Set();
126
+ for (let i = 0; i < probes && pairs.length < want && total > 0; i++) {
127
+ const slot = (i / probes + from) % 1;
128
+ const id = Math.floor(slot * total);
129
+ if (seen.has(id))
130
+ continue;
131
+ if (!store.has(id) || !store.hasNext(id))
132
+ continue;
133
+ if (store.contentLen(id, floorBytes) < floorBytes)
134
+ continue;
135
+ const pair = pairOf(ctx, id, 0);
136
+ if (pair) {
137
+ seen.add(id);
138
+ pairs.push(pair);
139
+ }
140
+ }
141
+ return {
142
+ pairs,
143
+ resolved: 0,
144
+ reached: pairs.length,
145
+ totalContexts: store.edgeSourceCount(),
146
+ browsed: true,
147
+ miss: pairs.length > 0 ? "matched" : "no-continuations",
148
+ };
149
+ }
@@ -162,14 +162,13 @@ export declare class GraphSearch {
162
162
  * recursive completion), and chooseNext (distributional-evidence edge
163
163
  * disambiguation when a recognised form has multiple continuations). */
164
164
  host: GraphSearchHost);
165
- /** The hub bound √N (AGENTS §2.8) — the ONE fan-out cap, stated here
166
- * rather than imported from `traverse.ts` because this module is
167
- * deliberately host-based (it holds a bare Store, never a MindContext).
168
- * That is the same write/read-side duplication convention canonical.ts's
169
- * header documents: if the formula changes it must change in BOTH places.
170
- * It is stated ONCE per side, though — the expression used to be spelled
171
- * out at three call sites here, one of them inside a per-item rules
172
- * generator, and they had already drifted on the `Math.max(2, …)` floor. */
165
+ /** The nodes the QUERY canonically names — the same identity the store's keys
166
+ * were written through. A byte-exact test is not enough: the query writes
167
+ * `Eiffel Tower country` and the deposited node is `eiffel tower country`, so
168
+ * a join that filters the query's own subject by RAW bytes re-admits it —
169
+ * measured: that is the trap's wrong answer (`The capital of Eiffel Tower
170
+ * country is Berlin.`). Cached by query identity, because the search is
171
+ * reused across responses. */
173
172
  private hubBound;
174
173
  /** Explore the Sema graph for the lightest cover of the query and return its
175
174
  * chosen spans left-to-right — WITH the derivation's total weight (the g
@@ -194,9 +194,17 @@ export class GraphSearch {
194
194
  this.maxGroup = maxGroup;
195
195
  this.host = host;
196
196
  }
197
- /** The hub bound √N (AGENTS §2.8) — the ONE fan-out cap, stated here
198
- * rather than imported from `traverse.ts` because this module is
199
- * deliberately host-based (it holds a bare Store, never a MindContext).
197
+ /** The nodes the QUERY canonically names — the same identity the store's keys
198
+ * were written through. A byte-exact test is not enough: the query writes
199
+ * `Eiffel Tower country` and the deposited node is `eiffel tower country`, so
200
+ * a join that filters the query's own subject by RAW bytes re-admits it —
201
+ * measured: that is the trap's wrong answer (`The capital of Eiffel Tower
202
+ * country is Berlin.`). Cached by query identity, because the search is
203
+ * reused across responses. */
204
+ /* * The hub bound √N (bounded-reads.md) — the ONE
205
+ * fan-out cap, stated here rather than imported from `traverse.ts` because
206
+ * this module is deliberately host-based (it holds a bare Store, never a
207
+ * MindContext).
200
208
  * That is the same write/read-side duplication convention canonical.ts's
201
209
  * header documents: if the formula changes it must change in BOTH places.
202
210
  * It is stated ONCE per side, though — the expression used to be spelled
@@ -460,7 +468,7 @@ export class GraphSearch {
460
468
  return this.coverRules(it, coversDone, coverableByStart);
461
469
  }
462
470
  if (it.kind === "form") {
463
- return this.formRules(it, conceptTarget, substitutions, nodeBytes);
471
+ return this.formRules(it, conceptTarget, substitutions, nodeBytes, queryLen);
464
472
  }
465
473
  return this.outRules(it, {
466
474
  W,
@@ -546,7 +554,7 @@ export class GraphSearch {
546
554
  }
547
555
  /** form(i,j,node,via): follow the graph out of `node`, or (in articulation)
548
556
  * emit its substitute voice directly. */
549
- *formRules(it, conceptTarget, substitutions, nodeBytes) {
557
+ *formRules(it, conceptTarget, substitutions, nodeBytes, queryLen) {
550
558
  // Articulation: emit voice bytes at the recognised span; the hop/concept/
551
559
  // emit chain is suppressed — the form contributes only its substitute.
552
560
  if (substitutions) {
@@ -580,7 +588,46 @@ export class GraphSearch {
580
588
  // guard then dead-ends it) with no way to reach the forward edge.
581
589
  // Forking offers every continuation as its own rule so the one that
582
590
  // genuinely advances (not a duplicate) is still reachable.
591
+ // A CHAIN HOP OFFERS ONLY WHAT THE QUESTION CAN PAY FOR.
592
+ //
593
+ // `hubBound` = √N is the READ cap — every read here stays inside it — but
594
+ // it is not an exploration bound: measured, a hub of degree 1083 sits
595
+ // BELOW √N = 1559, so a hop offered all 1083 continuations, the chart grew
596
+ // to 3113 outs for a two-word question, and since every out with an
597
+ // uncovered tail probes its tail's prefixes (measured: 16 885 canonical
598
+ // probes = 87% of that query's work, and its 270 MB peak / 256 MB OOM),
599
+ // the cost came from OFFERING rather than from reading.
600
+ //
601
+ // The bound is derived, not tuned: a derivation of L hops consumes ~L
602
+ // units of the question, so a hop cannot be paid for by offering more
603
+ // continuations than the question has units —
604
+ // `ceil(queryLen / W)`, floored at 2 for plurality. It is QUERY-sized
605
+ // (invariant 5: no per-query read grows with N) and it leaves `hubBound`
606
+ // and every read untouched.
607
+ // THE OFFER IS THE CORPUS'S OWN STRUCTURE, and the search pays for
608
+ // exploring it. There is no offer cap here any more: the traversal cap I
609
+ // had put on this hop was a short-circuit — it bounded what a hop could
610
+ // OFFER instead of charging for it — and it was not needed.
611
+ //
612
+ // MEASURED in the regime where it used to bite (`hubBound = ceil(√N)`
613
+ // GREATER than the hub's degree — reached in a fixture by choosing the
614
+ // degree below √N, so the trained store is not needed): with the cap the
615
+ // offer was 8/9/9 continuations at degrees 35/70/120; without it, 52/84/120
616
+ // — and the WORK is LINEAR in the degree, not quadratic: pushes 262/296/332,
617
+ // perceptions 530/592/757, while the PEAK is identical with and without the
618
+ // cap (218/415/689 MB against 215/410/662) because it is set by the store,
619
+ // not by the fan-out. What made this hop expensive was never the fan-out
620
+ // breadth: it was the per-offer work, two duplicate/oversized computations
621
+ // since removed (the per-offset canonical scan, and the tail scan now
622
+ // restricted to the fold's boundaries).
623
+ //
624
+ // The residual, stated: the trained store's hub (degree 1 083) is an
625
+ // EXTRAPOLATION from this linear shape, not a measurement.
583
626
  const nx = this.store.nextFirst(it.node, this.hubBound());
627
+ // Count what is OFFERED, not what was read: the evidence-preferred
628
+ // continuation is yielded too, even when it lies outside the cap.
629
+ if (this.host.meter)
630
+ this.host.meter.chainOffers += nx.length + 1;
584
631
  if (nx.length) {
585
632
  // The SAME evidence-weighted disambiguation the first hop uses
586
633
  // (below) identifies the most-corroborated continuation. Yielding
@@ -589,10 +636,14 @@ export class GraphSearch {
589
636
  // arrivals at an EQUAL cost (`cost < current`, strictly) — so
590
637
  // among same-depth sibling forks that tie in cost, the
591
638
  // evidence-backed edge wins deterministically, never by
592
- // exploration-order luck. `preferred`, when set, is necessarily
593
- // an element of `nx` (chooseNext reads the identical hub-bounded
594
- // set — see traverse.ts), so a plain skip-in-place suffices; no
595
- // second array need be allocated to reorder it to the front.
639
+ // exploration-order luck. `preferred` is NOT necessarily an element
640
+ // of `nx` any more: `nx` is the capped read above, while `chooseNext`
641
+ // reads its own hub-bounded set (see traverse.ts) — so the evidence
642
+ // pick may lie outside the cap, and it is still yielded FIRST on
643
+ // purpose. The cap bounds what the hop EXPLORES; it must never make
644
+ // the evidence-ranked continuation unreachable, or the derivation the
645
+ // corpus corroborates would lose to exploration order. Nothing is
646
+ // reallocated: `nx` is walked with a skip-in-place for the duplicate.
596
647
  const preferred = nx.length > 1
597
648
  ? this.host.chooseNext?.(it.node)
598
649
  : undefined;
@@ -779,12 +830,12 @@ export class GraphSearch {
779
830
  if (memo.has(node))
780
831
  return memo.get(node) ?? null;
781
832
  // Re-covering is how a PRODUCED node's bytes enter the search at all: the
782
- // cover machinery otherwise only ever sees the QUERY's spans. The recursion
833
+ // cover machinery otherwise only ever sees the QUERY's spans. The recursion
783
834
  // is allowed to nest — a chain IS nested completions — but it is bounded so
784
- // the work stays the ANSWER's (AGENTS §2.8): the stack below is the cycle
785
- // guard, only ACCEPTED completions recurse, and the nested solve decomposes
786
- // the form by its own shape instead of re-recognising the corpus's hub forms
787
- // inside it.
835
+ // the work stays the ANSWER's (bounded-reads.md): the stack below is the
836
+ // cycle guard, only ACCEPTED completions recurse, and the nested solve
837
+ // decomposes the form by its own shape instead of re-recognising the
838
+ // corpus's hub forms inside it.
788
839
  //
789
840
  // `recompleteOpen` IS the stack of the chain being built, so MEMBERSHIP is
790
841
  // the cycle guard: a node already open on this chain cannot re-enter it.
@@ -819,8 +870,46 @@ export class GraphSearch {
819
870
  // a deeper rewrite chain is made of.
820
871
  const rec = this.host.recogniseSpan(bytes);
821
872
  const kids = new Set(nrec.kids);
873
+ // THE NODE'S OWN KIDS ARE SITES BY STRUCTURE — recognition cannot be the
874
+ // only source. A produced composite is decomposed by ITS OWN SHAPE, and
875
+ // at hub scale the recognition of a produced span returns the WHOLE while
876
+ // deliberately suppressing its atoms (the off-boundary suppression), so
877
+ // the kid filter below would admit nothing at all and the chain would end
878
+ // at the intermediate composite. Measured on the chain
879
+ // `seed → "p q" → (p→r, q→s) → "r s" → "m n"`: below the flip it reaches
880
+ // "m n" with fuse+recompose, above it stops at "p q" — the trace shows
881
+ // `recognise("p q") ⇒ form "p q"` alone, no parts.
882
+ //
883
+ // Laying the node's kids out over its own bytes restores exactly the
884
+ // decomposition the node's tree already states; a kid recognition ALREADY
885
+ // offers is skipped, so below the flip the seed set is byte-identical to
886
+ // what it was. The filter's guarantee is untouched: nothing beyond the
887
+ // node's own kids may enter.
888
+ const recognised = rec.sites.filter((s) => kids.has(s.payload));
889
+ // DEDUPED BY SPAN, not by payload. A node whose kids repeat (`"abab"`
890
+ // folds as ["ab","ab"]) has TWO occurrences of the same node at different
891
+ // offsets; a payload-keyed set suppressed the structural site for BOTH, so
892
+ // the second occurrence had no site at all. The set now holds the spans
893
+ // recognition already covers, and a structural site is added exactly when
894
+ // nothing covers that PLACE — O(1) lookups, no extra reads.
895
+ const seenSpans = new Set(recognised.map((s) => `${s.start}:${s.end}`));
896
+ const structural = [];
897
+ {
898
+ let off = 0;
899
+ for (const k of nrec.kids) {
900
+ const len = this.store.bytesPrefix(k, ALL).length;
901
+ if (!seenSpans.has(`${off}:${off + len}`) && len > 0) {
902
+ structural.push({
903
+ start: off,
904
+ end: Math.min(off + len, bytes.length),
905
+ payload: k,
906
+ });
907
+ }
908
+ off += len;
909
+ }
910
+ }
822
911
  const solved = this.solve(bytes.length, {
823
- sites: rec.sites.filter((s) => kids.has(s.payload)),
912
+ sites: [...recognised, ...structural],
824
913
  leaves: rec.leaves,
825
914
  splits: rec.splits,
826
915
  starts: rec.starts,
@@ -891,6 +980,14 @@ export class GraphSearch {
891
980
  const tail = queryBytes.subarray(fact.j, queryLen);
892
981
  if (tail.length === 0)
893
982
  return;
983
+ // Report ONLY the invocations that could have joined: the search asks this
984
+ // rule for every finalized out with a node, which includes the one-byte
985
+ // outs the cover bridges with — measured, 68 refusals for a single
986
+ // 3-relation query, all of them letters. A form shorter than one window is
987
+ // not a fact a join could travel through, so it is not a refusal worth
988
+ // reporting; W is the same line the rest of the mind draws between a chance
989
+ // overlap and a form.
990
+ const reportable = fact.bytes.length >= this.maxGroup;
894
991
  // The entity candidates are the forms the fact's own bytes CONTAIN — the
895
992
  // same recogniser the query went through, so the evidence standard is the
896
993
  // query's. A byte atom is never a subject; the fact's own node is the span
@@ -899,10 +996,62 @@ export class GraphSearch {
899
996
  // LENDS it when it can (Mind does, with the response-scoped struct cache),
900
997
  // and a bare host falls back to the raw-store probe, so the search stays
901
998
  // host-based.
902
- const leading = this.host.recogniseSpan(fact.bytes).sites.filter((s) => s.payload >= 0 && s.payload !== fact.node &&
903
- (this.host.leadsSomewhere !== undefined
904
- ? this.host.leadsSomewhere(s.payload)
905
- : this.store.hasNext(s.payload) || this.store.hasHalo(s.payload)));
999
+ const factRec = this.host.recogniseSpan(fact.bytes);
1000
+ const leads = (id) => this.host.leadsSomewhere !== undefined
1001
+ ? this.host.leadsSomewhere(id)
1002
+ : this.store.hasNext(id) || this.store.hasHalo(id);
1003
+ // THE QUERY'S OWN SUBJECT, CANONICALLY. The filter used raw bytes and the
1004
+ // store's nodes are canonical, so `Eiffel Tower country` in the query did
1005
+ // not match the deposited `eiffel tower country` — measured, that is the
1006
+ // trap's wrong answer.
1007
+ //
1008
+ // TAKEN FROM THE RECOGNITION THE RESPONSE ALREADY COMPUTED — the host's
1009
+ // `recogniseSpan`, the same surface the rest of this search uses — not from
1010
+ // a second offset scan. `canonicalQueryNodes` re-derived, per byte offset,
1011
+ // what `recognise` had already resolved once per query (its memo is keyed by
1012
+ // content), and that scan was the largest single cost the DIANOT join added:
1013
+ // measured against the pre-change tree, the same fixture and the same test
1014
+ // were 21 s slower with the scan than without it. A recognised site IS a
1015
+ // canonical node of the query that can lead somewhere, which is exactly the
1016
+ // set this filter wants, and it costs nothing to read.
1017
+ const queryNodes = new Set((this.host.recogniseSpan?.(queryBytes)?.sites ?? []).map((s) => s.payload));
1018
+ // TWO SOURCES, ONE ADMISSION. The recognition of a STORED WHOLE returns the
1019
+ // whole and stops — measured: for `The director of Eva is Gustaf Molander.`
1020
+ // it yields exactly ONE site, the fact's own node — so the entity a join
1021
+ // exists for is never proposed. The canonical fold is the second source,
1022
+ // and the scan runs only for a FORM (≥ W: a one-byte out is not something to
1023
+ // join through, and running it per letter measured 20-26 s in test/99).
1024
+ const W = this.maxGroup;
1025
+ const proposed = new Map();
1026
+ // The SOURCE of each proposal travels with it: a refusal that names only the
1027
+ // bytes leaves the next reader guessing which path proposed them — three
1028
+ // attempts at the chained join were spent fixing paths that never produced
1029
+ // the offending candidate.
1030
+ const source = new Map();
1031
+ for (const s of factRec.sites) {
1032
+ if (s.payload >= 0 && leads(s.payload)) {
1033
+ proposed.set(s.payload, this.store.bytesPrefix(s.payload, ALL));
1034
+ source.set(s.payload, "recognised site");
1035
+ }
1036
+ }
1037
+ if (this.host.canonResolve !== undefined && fact.bytes.length >= W) {
1038
+ const canon = this.host.canonResolve.bind(this.host);
1039
+ for (let start = 0; start < fact.bytes.length; start++) {
1040
+ for (let end = fact.bytes.length; end - start >= W; end--) {
1041
+ const id = canon(fact.bytes.subarray(start, end));
1042
+ if (id === null)
1043
+ continue;
1044
+ if (leads(id)) {
1045
+ proposed.set(id, this.store.bytesPrefix(id, ALL));
1046
+ source.set(id, "canonical fold");
1047
+ }
1048
+ break; // the longest form at this offset wins
1049
+ }
1050
+ }
1051
+ }
1052
+ const leading = [...proposed]
1053
+ .filter(([payload]) => payload !== fact.node && !queryNodes.has(payload))
1054
+ .map(([payload, bytes]) => ({ payload, bytes }));
906
1055
  // …then prefer the entity the query did NOT name, and the MAXIMAL one. The
907
1056
  // join exists to reach the subject the query never wrote, so:
908
1057
  // • a candidate the query already contains is the query's OWN subject, and
@@ -913,30 +1062,93 @@ export class GraphSearch {
913
1062
  // introduces ("Timur" must not win over "Timur Bekmambetov").
914
1063
  // Byte work over bytes already read, and the pruning REMOVES the
915
1064
  // resolve()/nextFirst() probes these candidates would have paid.
1065
+ if (leading.length === 0) {
1066
+ if (this.host.meter)
1067
+ this.host.meter.joinNoEntity++;
1068
+ if (reportable) {
1069
+ // Report WHAT the recognition returned, not just that nothing led: the
1070
+ // count and the first few site texts are the difference between "the
1071
+ // fact was not recognised" and "it was recognised but nothing led".
1072
+ const seen = factRec.sites.slice(0, 3).map((s) => this.store.bytesPrefix(s.payload, ALL));
1073
+ this.host.reportSearch?.("deriveThroughMiss", [fact.bytes, tail, ...seen], `no entity inside the fact leads anywhere — ${factRec.sites.length} site(s) recognised inside it`);
1074
+ }
1075
+ }
916
1076
  const candidates = leading
917
- .map((s) => ({
918
- payload: s.payload,
919
- bytes: this.store.bytesPrefix(s.payload, ALL),
920
- }))
921
- .filter((c) => indexOf(queryBytes, c.bytes, 0) < 0)
922
1077
  .filter((c, _i, all) => !all.some((o) => o.bytes.length > c.bytes.length && indexOf(o.bytes, c.bytes, 0) >= 0));
923
1078
  for (const c of candidates) {
924
- const key = this.host.resolve(concat2(c.bytes, tail));
925
- if (key === null)
926
- continue;
927
- const nx = this.store.nextFirst(key, 1);
928
- if (nx.length === 0)
1079
+ // THE SHORTEST TAIL PREFIX WHOSE KEY ALSO LEADS SOMEWHERE.
1080
+ //
1081
+ // The conclusion covers only that prefix, so the rest of the tail stays
1082
+ // for the step after — which is what CHAINING is (a whole-tail key
1083
+ // consumed the whole remainder and made every join terminal: measured,
1084
+ // `joinFired=0` on a three-relation query).
1085
+ //
1086
+ // A KEY THAT RESOLVES IS NOT ENOUGH. Measured with a dry run of these
1087
+ // very primitives: for the candidate `Sweden` the first tail prefix that
1088
+ // resolves is `" "` — the key `Sweden ` (a trailing space) — and it leads
1089
+ // NOWHERE (`nextFirst` = 0), so accepting it refused the join while the
1090
+ // key that names the fact (`sweden capital`) sat one prefix further. The
1091
+ // loop therefore asks BOTH questions before accepting, and keeps looking
1092
+ // otherwise.
1093
+ //
1094
+ // EXACT FIRST, THEN CANONICAL: the corpus holds both identities.
1095
+ let key = null;
1096
+ let used = 0;
1097
+ // The continuation the accepted key leads to, carried out of the loop:
1098
+ // the loop already HAD to read it to accept the key (a key that leads
1099
+ // nowhere is not the relation), so re-reading it after the loop was a
1100
+ // duplicate read and a branch that could never be taken.
1101
+ let next = null;
1102
+ let keyBytes = c.bytes;
1103
+ // THE PREFIX ENDS ARE THE TAIL'S OWN FOLD BOUNDARIES, not every byte
1104
+ // length. The key is `entity + prefix`, and the prefix that names a
1105
+ // stored relation ends where the fold cuts: measured over four join-firing
1106
+ // queries, 5 of 5 accepted keys ended on a boundary (or the tail's end)
1107
+ // while the byte-by-byte scan spent 153 probes where 14 boundaries would
1108
+ // do. Same criterion — resolves AND leads — same shortest-first order, so
1109
+ // the answer is the same one the enumeration found; only the candidates
1110
+ // come from the structure instead of from the byte count. A host with no
1111
+ // boundary rule falls back to the enumeration.
1112
+ const cuts = this.host.contentCuts?.(tail);
1113
+ const ends = cuts && cuts.length > 0
1114
+ ? [...cuts.filter((c) => c > 0 && c < tail.length), tail.length]
1115
+ : Array.from({ length: tail.length }, (_, i) => i + 1);
1116
+ for (const len of ends) {
1117
+ keyBytes = concat2(c.bytes, tail.subarray(0, len));
1118
+ const k = this.host.resolve(keyBytes) ??
1119
+ this.host.canonResolve?.(keyBytes) ??
1120
+ null;
1121
+ if (k === null)
1122
+ continue;
1123
+ const nx = this.store.nextFirst(k, 1);
1124
+ if (nx.length === 0)
1125
+ continue;
1126
+ key = k;
1127
+ used = len;
1128
+ next = nx[0];
1129
+ break;
1130
+ }
1131
+ if (key === null) {
1132
+ if (this.host.meter)
1133
+ this.host.meter.joinNoKey++;
1134
+ if (reportable) {
1135
+ this.host.reportSearch?.("deriveThroughMiss", [c.bytes, tail, keyBytes], `no learnt key names this entity and tail together ` +
1136
+ `(candidate #${c.payload}, from the ${source.get(c.payload) ?? "unknown"} source)`);
1137
+ }
929
1138
  continue;
1139
+ }
1140
+ if (this.host.meter)
1141
+ this.host.meter.joinFired++;
930
1142
  yield {
931
1143
  premises: [fact],
932
1144
  conclusion: {
933
1145
  kind: "out",
934
1146
  i: fact.i,
935
- j: queryLen,
936
- bytes: this.store.bytesPrefix(nx[0], ALL),
1147
+ j: fact.j + used,
1148
+ bytes: this.store.bytesPrefix(next, ALL),
937
1149
  cover: true,
938
1150
  rec: true,
939
- node: nx[0],
1151
+ node: next,
940
1152
  throughFact: true,
941
1153
  },
942
1154
  cost: STEP,
@@ -1,5 +1,5 @@
1
1
  export { Mind } from "./mind.js";
2
- export type { Input, Response } from "./mind.js";
2
+ export type { CorpusTextPair, CorpusTextResult, Input, Response, } from "./mind.js";
3
3
  export type { ComputedSpan, ExtensionHost } from "./mind.js";
4
4
  export type { MechanismResult, PipelineMechanism, Precomputed, } from "./pipeline-mechanism.js";
5
5
  export type { InspectRationale, RationaleItem, RationaleStep, } from "./rationale.js";
@@ -7,3 +7,5 @@ export type { AnchorRejectionReason, ClimbConsensusData, ConsensusAnchorTrace, C
7
7
  export type { AncestorReach, AttentionRead, SaturationReason, SaturationStop, } from "./types.js";
8
8
  export type { DepositReport } from "./learning.js";
9
9
  export type { DecideGroundingData, NarrowDecisionData, Provenance, } from "./pipeline.js";
10
+ export { sampleCorpus, searchCorpus } from "./corpus.js";
11
+ export type { CorpusMiss, CorpusPair, CorpusResult } from "./corpus.js";
@@ -3,3 +3,4 @@
3
3
  // Re-exports the Mind class and all public types that were previously
4
4
  // exported from mind/mind.ts directly.
5
5
  export { Mind } from "./mind.js";
6
+ export { sampleCorpus, searchCorpus } from "./corpus.js";
@@ -97,7 +97,7 @@ export declare function cachedRead(ctx: MindContext, cache: WalkCache | null, id
97
97
  * is legitimately reached across many containing structures. Half the
98
98
  * successful junctions would be lost.
99
99
  *
100
- * REFUTED EARLY-STOP (side-cone exhaustion, §2.17's "real saturation"):
100
+ * REFUTED EARLY-STOP (side-cone exhaustion, saturation.md's "real saturation"):
101
101
  * stopping the walk the moment ONE side's upward cone is emptied is wrong,
102
102
  * in both a hub-guarded form and a hub-flagged form. The junction test is
103
103
  * a BYTE containment over the UNION of the two cones, and a junction can be
@@ -135,7 +135,7 @@ function cachedContainers(ctx, cache, id, limit) {
135
135
  * is legitimately reached across many containing structures. Half the
136
136
  * successful junctions would be lost.
137
137
  *
138
- * REFUTED EARLY-STOP (side-cone exhaustion, §2.17's "real saturation"):
138
+ * REFUTED EARLY-STOP (side-cone exhaustion, saturation.md's "real saturation"):
139
139
  * stopping the walk the moment ONE side's upward cone is emptied is wrong,
140
140
  * in both a hub-guarded form and a hub-flagged form. The junction test is
141
141
  * a BYTE containment over the UNION of the two cones, and a junction can be
@@ -186,13 +186,13 @@ unordered = false) {
186
186
  d: 0,
187
187
  }));
188
188
  while (stack.length > 0 && out.length < bound) {
189
- // BUDGET EXHAUSTION IS AN ABSTENTION, AND IT MUST BE VISIBLE (§2.13). The
190
- // walk stops with work still on the stack, the caller reads "no container"
191
- // and falls through to a lower ladder rung — indistinguishable, from the
192
- // outside, from a walk that looked everywhere and found nothing. With a
193
- // SHARED budget (cross-region's one k·W allowance per tier) an EARLIER
194
- // pair can drain it, so a later pair's exact tier may never run at all;
195
- // this counter is the only thing that says so.
189
+ // BUDGET EXHAUSTION IS AN ABSTENTION, AND IT MUST BE VISIBLE
190
+ // (INVARIANTS.md). The walk stops with work still on the stack, the caller
191
+ // reads "no container" and falls through to a lower ladder rung —
192
+ // indistinguishable, from the outside, from a walk that looked everywhere
193
+ // and found nothing. With a SHARED budget (cross-region's one k·W allowance
194
+ // per tier) an EARLIER pair can drain it, so a later pair's exact tier may
195
+ // never run at all; this counter is the only thing that says so.
196
196
  if (b.n-- <= 0) {
197
197
  if (ctx.meter)
198
198
  ctx.meter.junctionBudgetExhausted++;