@hviana/sema 0.8.1 → 0.8.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (114) hide show
  1. package/AGENTS.md +29 -29
  2. package/TRADEMARKS.md +0 -1
  3. package/dist/src/config.d.ts +28 -0
  4. package/dist/src/config.js +20 -0
  5. package/dist/src/geometry.d.ts +21 -0
  6. package/dist/src/geometry.js +21 -0
  7. package/dist/src/meter.d.ts +76 -0
  8. package/dist/src/meter.js +95 -0
  9. package/dist/src/mind/attention.d.ts +4 -0
  10. package/dist/src/mind/attention.js +165 -16
  11. package/dist/src/mind/canonical.d.ts +16 -0
  12. package/dist/src/mind/canonical.js +41 -0
  13. package/dist/src/mind/corpus.d.ts +40 -0
  14. package/dist/src/mind/corpus.js +149 -0
  15. package/dist/src/mind/graph-search.d.ts +7 -0
  16. package/dist/src/mind/graph-search.js +254 -24
  17. package/dist/src/mind/index.d.ts +3 -1
  18. package/dist/src/mind/index.js +1 -0
  19. package/dist/src/mind/match.d.ts +9 -4
  20. package/dist/src/mind/match.js +147 -61
  21. package/dist/src/mind/mechanisms/cast.js +19 -3
  22. package/dist/src/mind/mechanisms/confluence.js +24 -0
  23. package/dist/src/mind/mechanisms/cover.js +6 -0
  24. package/dist/src/mind/mechanisms/recall.js +32 -4
  25. package/dist/src/mind/mind.d.ts +57 -0
  26. package/dist/src/mind/mind.js +72 -1
  27. package/dist/src/mind/pipeline-mechanism.d.ts +7 -0
  28. package/dist/src/mind/pipeline.js +66 -20
  29. package/dist/src/mind/primitives.js +9 -1
  30. package/dist/src/mind/rationale.d.ts +28 -1
  31. package/dist/src/mind/rationale.js +22 -1
  32. package/dist/src/mind/reasoning.d.ts +25 -3
  33. package/dist/src/mind/reasoning.js +125 -20
  34. package/dist/src/mind/recognition.js +4 -8
  35. package/dist/src/mind/resonance.js +20 -1
  36. package/dist/src/mind/trace.js +1 -0
  37. package/dist/src/mind/traverse.js +15 -3
  38. package/dist/src/mind/types.d.ts +49 -4
  39. package/docs/INVARIANTS.md +2 -2
  40. package/docs/architecture/bounded-reads.md +1 -1
  41. package/docs/architecture/commonality.md +2 -2
  42. package/docs/architecture/cost-model.md +2 -2
  43. package/docs/architecture/determinism.md +7 -7
  44. package/docs/architecture/match-project.md +2 -3
  45. package/docs/architecture/mechanism-market.md +10 -10
  46. package/docs/architecture/meter.md +5 -5
  47. package/docs/architecture/store.md +3 -3
  48. package/docs/failures/tempting-but-wrong.md +34 -6
  49. package/docs/harness/gates.md +2 -2
  50. package/docs/mechanisms/cast.md +2 -2
  51. package/docs/mechanisms/cover.md +2 -3
  52. package/docs/mechanisms/extraction.md +7 -7
  53. package/docs/mechanisms/recall.md +8 -9
  54. package/jsr.json +1 -1
  55. package/package.json +1 -1
  56. package/src/alu/README.md +11 -12
  57. package/src/config.ts +48 -0
  58. package/src/geometry.ts +21 -0
  59. package/src/meter.ts +98 -0
  60. package/src/mind/attention.ts +167 -16
  61. package/src/mind/canonical.ts +43 -0
  62. package/src/mind/corpus.ts +202 -0
  63. package/src/mind/graph-search.ts +277 -23
  64. package/src/mind/index.ts +8 -1
  65. package/src/mind/match.ts +148 -57
  66. package/src/mind/mechanisms/cast.ts +20 -2
  67. package/src/mind/mechanisms/confluence.ts +24 -0
  68. package/src/mind/mechanisms/cover.ts +5 -0
  69. package/src/mind/mechanisms/recall.ts +32 -4
  70. package/src/mind/mind.ts +125 -0
  71. package/src/mind/pipeline-mechanism.ts +7 -0
  72. package/src/mind/pipeline.ts +79 -22
  73. package/src/mind/primitives.ts +9 -1
  74. package/src/mind/rationale.ts +35 -1
  75. package/src/mind/reasoning.ts +145 -13
  76. package/src/mind/recognition.ts +4 -8
  77. package/src/mind/resonance.ts +19 -1
  78. package/src/mind/trace.ts +1 -0
  79. package/src/mind/traverse.ts +16 -6
  80. package/src/mind/types.ts +53 -4
  81. package/test/100-complete-grounding-trace.test.mjs +109 -0
  82. package/test/101-alignment-gap-bound.test.mjs +106 -0
  83. package/test/102-production-composes-at-scale.test.mjs +110 -0
  84. package/test/103-alignment-gap-budget.test.mjs +89 -0
  85. package/test/104-composition-is-reported.test.mjs +90 -0
  86. package/test/105-derive-through-reports-its-refusal.test.mjs +137 -0
  87. package/test/106-the-join-fires.test.mjs +94 -0
  88. package/test/107-the-join-is-counted.test.mjs +81 -0
  89. package/test/108-the-join-chains.test.mjs +78 -0
  90. package/test/109-the-pivot-is-counted.test.mjs +60 -0
  91. package/test/110-the-reasoner-stops-when-the-question-is-answered.test.mjs +91 -0
  92. package/test/111-the-cover-assembly-is-counted.test.mjs +74 -0
  93. package/test/112-the-exploration-does-not-grow-with-the-hub.test.mjs +89 -0
  94. package/test/113-the-rationale-payload-is-bounded.test.mjs +84 -0
  95. package/test/114-alignment-budget-is-per-sweep.test.mjs +93 -0
  96. package/test/116-the-extension-is-gated-by-the-pipelines-own-remainder.test.mjs +100 -0
  97. package/test/117-corpus-search.test.mjs +171 -0
  98. package/test/118-the-join-reaches-a-key-off-the-cut.test.mjs +74 -0
  99. package/test/119-the-work-does-not-grow-with-the-corpus.test.mjs +122 -0
  100. package/test/120-composition-is-consequence.test.mjs +132 -0
  101. package/test/121-the-extension-does-not-grow-with-the-corpus.test.mjs +128 -0
  102. package/test/122-the-climb-search-does-not-grow-with-the-corpus.test.mjs +117 -0
  103. package/test/123-the-paired-formulas-agree.test.mjs +90 -0
  104. package/test/125-the-post-grounding-branch-publishes-its-operand.test.mjs +51 -0
  105. package/test/126-the-pipeline-does-not-name-mechanisms.test.mjs +42 -0
  106. package/test/128-the-leads-somewhere-pair-agrees.test.mjs +83 -0
  107. package/test/129-the-trace-payload-shape.test.mjs +164 -0
  108. package/test/14-scaling.test.mjs +10 -7
  109. package/test/32-confluence.test.mjs +68 -0
  110. package/test/38-reason-restate-guard.test.mjs +8 -2
  111. package/test/43-cast-analog-seat.test.mjs +10 -0
  112. package/test/55-cost-meter.test.mjs +859 -0
  113. package/test/76-reference-binding.test.mjs +6 -1
  114. package/test/89-completion-recursion.test.mjs +30 -5
@@ -194,6 +194,13 @@ export class GraphSearch {
194
194
  this.maxGroup = maxGroup;
195
195
  this.host = host;
196
196
  }
197
+ /** The nodes the QUERY canonically names — the same identity the store's keys
198
+ * were written through. A byte-exact test is not enough: the query writes
199
+ * `Eiffel Tower country` and the deposited node is `eiffel tower country`, so
200
+ * a join that filters the query's own subject by RAW bytes re-admits it —
201
+ * measured: that is the trap's wrong answer (`The capital of Eiffel Tower
202
+ * country is Berlin.`). Cached by query identity, because the search is
203
+ * reused across responses. */
197
204
  /* * The hub bound √N (bounded-reads.md) — the ONE
198
205
  * fan-out cap, stated here rather than imported from `traverse.ts` because
199
206
  * this module is deliberately host-based (it holds a bare Store, never a
@@ -461,7 +468,7 @@ export class GraphSearch {
461
468
  return this.coverRules(it, coversDone, coverableByStart);
462
469
  }
463
470
  if (it.kind === "form") {
464
- return this.formRules(it, conceptTarget, substitutions, nodeBytes);
471
+ return this.formRules(it, conceptTarget, substitutions, nodeBytes, queryLen);
465
472
  }
466
473
  return this.outRules(it, {
467
474
  W,
@@ -547,7 +554,7 @@ export class GraphSearch {
547
554
  }
548
555
  /** form(i,j,node,via): follow the graph out of `node`, or (in articulation)
549
556
  * emit its substitute voice directly. */
550
- *formRules(it, conceptTarget, substitutions, nodeBytes) {
557
+ *formRules(it, conceptTarget, substitutions, nodeBytes, queryLen) {
551
558
  // Articulation: emit voice bytes at the recognised span; the hop/concept/
552
559
  // emit chain is suppressed — the form contributes only its substitute.
553
560
  if (substitutions) {
@@ -581,7 +588,46 @@ export class GraphSearch {
581
588
  // guard then dead-ends it) with no way to reach the forward edge.
582
589
  // Forking offers every continuation as its own rule so the one that
583
590
  // genuinely advances (not a duplicate) is still reachable.
591
+ // A CHAIN HOP OFFERS ONLY WHAT THE QUESTION CAN PAY FOR.
592
+ //
593
+ // `hubBound` = √N is the READ cap — every read here stays inside it — but
594
+ // it is not an exploration bound: measured, a hub of degree 1083 sits
595
+ // BELOW √N = 1559, so a hop offered all 1083 continuations, the chart grew
596
+ // to 3113 outs for a two-word question, and since every out with an
597
+ // uncovered tail probes its tail's prefixes (measured: 16 885 canonical
598
+ // probes = 87% of that query's work, and its 270 MB peak / 256 MB OOM),
599
+ // the cost came from OFFERING rather than from reading.
600
+ //
601
+ // The bound is derived, not tuned: a derivation of L hops consumes ~L
602
+ // units of the question, so a hop cannot be paid for by offering more
603
+ // continuations than the question has units —
604
+ // `ceil(queryLen / W)`, floored at 2 for plurality. It is QUERY-sized
605
+ // (invariant 5: no per-query read grows with N) and it leaves `hubBound`
606
+ // and every read untouched.
607
+ // THE OFFER IS THE CORPUS'S OWN STRUCTURE, and the search pays for
608
+ // exploring it. There is no offer cap here any more: the traversal cap I
609
+ // had put on this hop was a short-circuit — it bounded what a hop could
610
+ // OFFER instead of charging for it — and it was not needed.
611
+ //
612
+ // MEASURED in the regime where it used to bite (`hubBound = ceil(√N)`
613
+ // GREATER than the hub's degree — reached in a fixture by choosing the
614
+ // degree below √N, so the trained store is not needed): with the cap the
615
+ // offer was 8/9/9 continuations at degrees 35/70/120; without it, 52/84/120
616
+ // — and the WORK is LINEAR in the degree, not quadratic: pushes 262/296/332,
617
+ // perceptions 530/592/757, while the PEAK is identical with and without the
618
+ // cap (218/415/689 MB against 215/410/662) because it is set by the store,
619
+ // not by the fan-out. What made this hop expensive was never the fan-out
620
+ // breadth: it was the per-offer work, two duplicate/oversized computations
621
+ // since removed (the per-offset canonical scan, and the tail scan now
622
+ // restricted to the fold's boundaries).
623
+ //
624
+ // The residual, stated: the trained store's hub (degree 1 083) is an
625
+ // EXTRAPOLATION from this linear shape, not a measurement.
584
626
  const nx = this.store.nextFirst(it.node, this.hubBound());
627
+ // Count what is OFFERED, not what was read: the evidence-preferred
628
+ // continuation is yielded too, even when it lies outside the cap.
629
+ if (this.host.meter)
630
+ this.host.meter.chainOffers += nx.length + 1;
585
631
  if (nx.length) {
586
632
  // The SAME evidence-weighted disambiguation the first hop uses
587
633
  // (below) identifies the most-corroborated continuation. Yielding
@@ -590,10 +636,14 @@ export class GraphSearch {
590
636
  // arrivals at an EQUAL cost (`cost < current`, strictly) — so
591
637
  // among same-depth sibling forks that tie in cost, the
592
638
  // evidence-backed edge wins deterministically, never by
593
- // exploration-order luck. `preferred`, when set, is necessarily
594
- // an element of `nx` (chooseNext reads the identical hub-bounded
595
- // set — see traverse.ts), so a plain skip-in-place suffices; no
596
- // second array need be allocated to reorder it to the front.
639
+ // exploration-order luck. `preferred` is NOT necessarily an element
640
+ // of `nx` any more: `nx` is the capped read above, while `chooseNext`
641
+ // reads its own hub-bounded set (see traverse.ts) — so the evidence
642
+ // pick may lie outside the cap, and it is still yielded FIRST on
643
+ // purpose. The cap bounds what the hop EXPLORES; it must never make
644
+ // the evidence-ranked continuation unreachable, or the derivation the
645
+ // corpus corroborates would lose to exploration order. Nothing is
646
+ // reallocated: `nx` is walked with a skip-in-place for the duplicate.
597
647
  const preferred = nx.length > 1
598
648
  ? this.host.chooseNext?.(it.node)
599
649
  : undefined;
@@ -818,10 +868,50 @@ export class GraphSearch {
818
868
  // concepts/connectors either (those need the caller's async
819
869
  // pre-resolution) — the recursion follows edges and fusion, which is what
820
870
  // a deeper rewrite chain is made of.
871
+ if (this.host.meter)
872
+ this.host.meter.recompletes++;
821
873
  const rec = this.host.recogniseSpan(bytes);
822
874
  const kids = new Set(nrec.kids);
875
+ // THE NODE'S OWN KIDS ARE SITES BY STRUCTURE — recognition cannot be the
876
+ // only source. A produced composite is decomposed by ITS OWN SHAPE, and
877
+ // at hub scale the recognition of a produced span returns the WHOLE while
878
+ // deliberately suppressing its atoms (the off-boundary suppression), so
879
+ // the kid filter below would admit nothing at all and the chain would end
880
+ // at the intermediate composite. Measured on the chain
881
+ // `seed → "p q" → (p→r, q→s) → "r s" → "m n"`: below the flip it reaches
882
+ // "m n" with fuse+recompose, above it stops at "p q" — the trace shows
883
+ // `recognise("p q") ⇒ form "p q"` alone, no parts.
884
+ //
885
+ // Laying the node's kids out over its own bytes restores exactly the
886
+ // decomposition the node's tree already states; a kid recognition ALREADY
887
+ // offers is skipped, so below the flip the seed set is byte-identical to
888
+ // what it was. The filter's guarantee is untouched: nothing beyond the
889
+ // node's own kids may enter.
890
+ const recognised = rec.sites.filter((s) => kids.has(s.payload));
891
+ // DEDUPED BY SPAN, not by payload. A node whose kids repeat (`"abab"`
892
+ // folds as ["ab","ab"]) has TWO occurrences of the same node at different
893
+ // offsets; a payload-keyed set suppressed the structural site for BOTH, so
894
+ // the second occurrence had no site at all. The set now holds the spans
895
+ // recognition already covers, and a structural site is added exactly when
896
+ // nothing covers that PLACE — O(1) lookups, no extra reads.
897
+ const seenSpans = new Set(recognised.map((s) => `${s.start}:${s.end}`));
898
+ const structural = [];
899
+ {
900
+ let off = 0;
901
+ for (const k of nrec.kids) {
902
+ const len = this.store.bytesPrefix(k, ALL).length;
903
+ if (!seenSpans.has(`${off}:${off + len}`) && len > 0) {
904
+ structural.push({
905
+ start: off,
906
+ end: Math.min(off + len, bytes.length),
907
+ payload: k,
908
+ });
909
+ }
910
+ off += len;
911
+ }
912
+ }
823
913
  const solved = this.solve(bytes.length, {
824
- sites: rec.sites.filter((s) => kids.has(s.payload)),
914
+ sites: [...recognised, ...structural],
825
915
  leaves: rec.leaves,
826
916
  splits: rec.splits,
827
917
  starts: rec.starts,
@@ -892,6 +982,14 @@ export class GraphSearch {
892
982
  const tail = queryBytes.subarray(fact.j, queryLen);
893
983
  if (tail.length === 0)
894
984
  return;
985
+ // Report ONLY the invocations that could have joined: the search asks this
986
+ // rule for every finalized out with a node, which includes the one-byte
987
+ // outs the cover bridges with — measured, 68 refusals for a single
988
+ // 3-relation query, all of them letters. A form shorter than one window is
989
+ // not a fact a join could travel through, so it is not a refusal worth
990
+ // reporting; W is the same line the rest of the mind draws between a chance
991
+ // overlap and a form.
992
+ const reportable = fact.bytes.length >= this.maxGroup;
895
993
  // The entity candidates are the forms the fact's own bytes CONTAIN — the
896
994
  // same recogniser the query went through, so the evidence standard is the
897
995
  // query's. A byte atom is never a subject; the fact's own node is the span
@@ -900,10 +998,62 @@ export class GraphSearch {
900
998
  // LENDS it when it can (Mind does, with the response-scoped struct cache),
901
999
  // and a bare host falls back to the raw-store probe, so the search stays
902
1000
  // host-based.
903
- const leading = this.host.recogniseSpan(fact.bytes).sites.filter((s) => s.payload >= 0 && s.payload !== fact.node &&
904
- (this.host.leadsSomewhere !== undefined
905
- ? this.host.leadsSomewhere(s.payload)
906
- : this.store.hasNext(s.payload) || this.store.hasHalo(s.payload)));
1001
+ const factRec = this.host.recogniseSpan(fact.bytes);
1002
+ const leads = (id) => this.host.leadsSomewhere !== undefined
1003
+ ? this.host.leadsSomewhere(id)
1004
+ : this.store.hasNext(id) || this.store.hasHalo(id);
1005
+ // THE QUERY'S OWN SUBJECT, CANONICALLY. The filter used raw bytes and the
1006
+ // store's nodes are canonical, so `Eiffel Tower country` in the query did
1007
+ // not match the deposited `eiffel tower country` — measured, that is the
1008
+ // trap's wrong answer.
1009
+ //
1010
+ // TAKEN FROM THE RECOGNITION THE RESPONSE ALREADY COMPUTED — the host's
1011
+ // `recogniseSpan`, the same surface the rest of this search uses — not from
1012
+ // a second offset scan. `canonicalQueryNodes` re-derived, per byte offset,
1013
+ // what `recognise` had already resolved once per query (its memo is keyed by
1014
+ // content), and that scan was the largest single cost the DIANOT join added:
1015
+ // measured against the pre-change tree, the same fixture and the same test
1016
+ // were 21 s slower with the scan than without it. A recognised site IS a
1017
+ // canonical node of the query that can lead somewhere, which is exactly the
1018
+ // set this filter wants, and it costs nothing to read.
1019
+ const queryNodes = new Set((this.host.recogniseSpan?.(queryBytes)?.sites ?? []).map((s) => s.payload));
1020
+ // TWO SOURCES, ONE ADMISSION. The recognition of a STORED WHOLE returns the
1021
+ // whole and stops — measured: for `The director of Eva is Gustaf Molander.`
1022
+ // it yields exactly ONE site, the fact's own node — so the entity a join
1023
+ // exists for is never proposed. The canonical fold is the second source,
1024
+ // and the scan runs only for a FORM (≥ W: a one-byte out is not something to
1025
+ // join through, and running it per letter measured 20-26 s in test/99).
1026
+ const W = this.maxGroup;
1027
+ const proposed = new Map();
1028
+ // The SOURCE of each proposal travels with it: a refusal that names only the
1029
+ // bytes leaves the next reader guessing which path proposed them — three
1030
+ // attempts at the chained join were spent fixing paths that never produced
1031
+ // the offending candidate.
1032
+ const source = new Map();
1033
+ for (const s of factRec.sites) {
1034
+ if (s.payload >= 0 && leads(s.payload)) {
1035
+ proposed.set(s.payload, this.store.bytesPrefix(s.payload, ALL));
1036
+ source.set(s.payload, "recognised site");
1037
+ }
1038
+ }
1039
+ if (this.host.canonResolve !== undefined && fact.bytes.length >= W) {
1040
+ const canon = this.host.canonResolve.bind(this.host);
1041
+ for (let start = 0; start < fact.bytes.length; start++) {
1042
+ for (let end = fact.bytes.length; end - start >= W; end--) {
1043
+ const id = canon(fact.bytes.subarray(start, end));
1044
+ if (id === null)
1045
+ continue;
1046
+ if (leads(id)) {
1047
+ proposed.set(id, this.store.bytesPrefix(id, ALL));
1048
+ source.set(id, "canonical fold");
1049
+ }
1050
+ break; // the longest form at this offset wins
1051
+ }
1052
+ }
1053
+ }
1054
+ const leading = [...proposed]
1055
+ .filter(([payload]) => payload !== fact.node && !queryNodes.has(payload))
1056
+ .map(([payload, bytes]) => ({ payload, bytes }));
907
1057
  // …then prefer the entity the query did NOT name, and the MAXIMAL one. The
908
1058
  // join exists to reach the subject the query never wrote, so:
909
1059
  // • a candidate the query already contains is the query's OWN subject, and
@@ -914,30 +1064,110 @@ export class GraphSearch {
914
1064
  // introduces ("Timur" must not win over "Timur Bekmambetov").
915
1065
  // Byte work over bytes already read, and the pruning REMOVES the
916
1066
  // resolve()/nextFirst() probes these candidates would have paid.
1067
+ if (leading.length === 0) {
1068
+ if (this.host.meter)
1069
+ this.host.meter.joinNoEntity++;
1070
+ if (reportable) {
1071
+ // Report WHAT the recognition returned, not just that nothing led: the
1072
+ // count and the first few site texts are the difference between "the
1073
+ // fact was not recognised" and "it was recognised but nothing led".
1074
+ const seen = factRec.sites.slice(0, 3).map((s) => this.store.bytesPrefix(s.payload, ALL));
1075
+ this.host.reportSearch?.("deriveThroughMiss", [fact.bytes, tail, ...seen], `no entity inside the fact leads anywhere — ${factRec.sites.length} site(s) recognised inside it`);
1076
+ }
1077
+ }
917
1078
  const candidates = leading
918
- .map((s) => ({
919
- payload: s.payload,
920
- bytes: this.store.bytesPrefix(s.payload, ALL),
921
- }))
922
- .filter((c) => indexOf(queryBytes, c.bytes, 0) < 0)
923
1079
  .filter((c, _i, all) => !all.some((o) => o.bytes.length > c.bytes.length && indexOf(o.bytes, c.bytes, 0) >= 0));
924
1080
  for (const c of candidates) {
925
- const key = this.host.resolve(concat2(c.bytes, tail));
926
- if (key === null)
927
- continue;
928
- const nx = this.store.nextFirst(key, 1);
929
- if (nx.length === 0)
1081
+ // THE SHORTEST TAIL PREFIX WHOSE KEY ALSO LEADS SOMEWHERE.
1082
+ //
1083
+ // The conclusion covers only that prefix, so the rest of the tail stays
1084
+ // for the step after — which is what CHAINING is (a whole-tail key
1085
+ // consumed the whole remainder and made every join terminal: measured,
1086
+ // `joinFired=0` on a three-relation query).
1087
+ //
1088
+ // A KEY THAT RESOLVES IS NOT ENOUGH. Measured with a dry run of these
1089
+ // very primitives: for the candidate `Sweden` the first tail prefix that
1090
+ // resolves is `" "` — the key `Sweden ` (a trailing space) — and it leads
1091
+ // NOWHERE (`nextFirst` = 0), so accepting it refused the join while the
1092
+ // key that names the fact (`sweden capital`) sat one prefix further. The
1093
+ // loop therefore asks BOTH questions before accepting, and keeps looking
1094
+ // otherwise.
1095
+ //
1096
+ // EXACT FIRST, THEN CANONICAL: the corpus holds both identities.
1097
+ let key = null;
1098
+ let used = 0;
1099
+ // The continuation the accepted key leads to, carried out of the loop:
1100
+ // the loop already HAD to read it to accept the key (a key that leads
1101
+ // nowhere is not the relation), so re-reading it after the loop was a
1102
+ // duplicate read and a branch that could never be taken.
1103
+ let next = null;
1104
+ let keyBytes = c.bytes;
1105
+ // THE CANDIDATE ENDS ARE THE PREFIXES THAT ARE STORED NODES, ASCENDING.
1106
+ // The key is `entity + prefix`, and it names a relation exactly when that
1107
+ // concatenation IS a node — so the ends come from a content-addressed
1108
+ // probe per offset (the host's `contentKeyEnds`, the learning path's own
1109
+ // mechanism: one leaf walk plus one `findBranch` per offset, no `resolve`),
1110
+ // never from the fold's boundaries. A boundary is not a proxy: a stored
1111
+ // member's end is the end of ITS OWN stream, and the fold never cuts at a
1112
+ // stream's end — measured, "stockholm mayor" exists, leads on to the mayor
1113
+ // fact, and its boundary 6 sits in neither the tail's cuts ([4,7]) nor the
1114
+ // concatenation's. Filtering the scan by "is this a node?" cannot change
1115
+ // the winner: a position that is not a node cannot resolve, so skipping it
1116
+ // is invisible; and the order stays SHORTEST FIRST, which is a semantic
1117
+ // law, not an optimisation (test/106, test/108 pin it).
1118
+ //
1119
+ // A host that cannot answer falls back to every prefix: exact and
1120
+ // complete, at a `resolve` per offset. A host that CAN answer is
1121
+ // authoritative even when it answers "none" — if no prefix is a node then
1122
+ // no key exists to resolve, so enumerating would only pay nulls. (A key
1123
+ // reachable through the CANONICAL equivalence alone and ending off every
1124
+ // node end is therefore not tried here; that dimension is unreachable on
1125
+ // this path by construction and is not part of the exact-key law.)
1126
+ const ends = this.host.contentKeyEnds?.(c.bytes, tail);
1127
+ const candidateEnds = function* () {
1128
+ if (ends !== undefined) {
1129
+ yield* ends;
1130
+ return;
1131
+ }
1132
+ for (let p = 1; p <= tail.length; p++)
1133
+ yield p;
1134
+ };
1135
+ for (const len of candidateEnds()) {
1136
+ keyBytes = concat2(c.bytes, tail.subarray(0, len));
1137
+ const k = this.host.resolve(keyBytes) ??
1138
+ this.host.canonResolve?.(keyBytes) ??
1139
+ null;
1140
+ if (k === null)
1141
+ continue;
1142
+ const nx = this.store.nextFirst(k, 1);
1143
+ if (nx.length === 0)
1144
+ continue;
1145
+ key = k;
1146
+ used = len;
1147
+ next = nx[0];
1148
+ break;
1149
+ }
1150
+ if (key === null) {
1151
+ if (this.host.meter)
1152
+ this.host.meter.joinNoKey++;
1153
+ if (reportable) {
1154
+ this.host.reportSearch?.("deriveThroughMiss", [c.bytes, tail, keyBytes], `no learnt key names this entity and tail together ` +
1155
+ `(candidate #${c.payload}, from the ${source.get(c.payload) ?? "unknown"} source)`);
1156
+ }
930
1157
  continue;
1158
+ }
1159
+ if (this.host.meter)
1160
+ this.host.meter.joinFired++;
931
1161
  yield {
932
1162
  premises: [fact],
933
1163
  conclusion: {
934
1164
  kind: "out",
935
1165
  i: fact.i,
936
- j: queryLen,
937
- bytes: this.store.bytesPrefix(nx[0], ALL),
1166
+ j: fact.j + used,
1167
+ bytes: this.store.bytesPrefix(next, ALL),
938
1168
  cover: true,
939
1169
  rec: true,
940
- node: nx[0],
1170
+ node: next,
941
1171
  throughFact: true,
942
1172
  },
943
1173
  cost: STEP,
@@ -1,5 +1,5 @@
1
1
  export { Mind } from "./mind.js";
2
- export type { Input, Response } from "./mind.js";
2
+ export type { CorpusTextPair, CorpusTextResult, Input, Response, } from "./mind.js";
3
3
  export type { ComputedSpan, ExtensionHost } from "./mind.js";
4
4
  export type { MechanismResult, PipelineMechanism, Precomputed, } from "./pipeline-mechanism.js";
5
5
  export type { InspectRationale, RationaleItem, RationaleStep, } from "./rationale.js";
@@ -7,3 +7,5 @@ export type { AnchorRejectionReason, ClimbConsensusData, ConsensusAnchorTrace, C
7
7
  export type { AncestorReach, AttentionRead, SaturationReason, SaturationStop, } from "./types.js";
8
8
  export type { DepositReport } from "./learning.js";
9
9
  export type { DecideGroundingData, NarrowDecisionData, Provenance, } from "./pipeline.js";
10
+ export { sampleCorpus, searchCorpus } from "./corpus.js";
11
+ export type { CorpusMiss, CorpusPair, CorpusResult } from "./corpus.js";
@@ -3,3 +3,4 @@
3
3
  // Re-exports the Mind class and all public types that were previously
4
4
  // exported from mind/mind.ts directly.
5
5
  export { Mind } from "./mind.js";
6
+ export { sampleCorpus, searchCorpus } from "./corpus.js";
@@ -78,9 +78,14 @@ export interface AlignGap {
78
78
  }
79
79
  /** Extend a seed match (query offset qo ↔ candidate offset co) to its maximal
80
80
  * common run, then walk outward in both directions collecting further common
81
- * runs of at least W bytes across bounded mismatch gaps (each side ≤
82
- * chainReach). Returns the matched query spans and the mismatch pairs
83
- * between consecutive runs.
81
+ * runs of at least W bytes across mismatch gaps. Each gap's LENGTH is the
82
+ * pair's own extent (a gap cannot be longer than the bytes it spans) and the
83
+ * sweep's WORK is proportional to the bytes a run spans (the context's windows
84
+ * are indexed once, then the query's are walked) — the arity bound
85
+ * (`chainReach`) used to cap BOTH, and truncated every learned frame whose
86
+ * slot was longer. Each sweep owns its own budget, so an exhausted right
87
+ * sweep never starves the left one. Returns the matched query spans and the
88
+ * mismatch pairs between consecutive runs.
84
89
  *
85
90
  * This is the SEEDED aligner, distinct from {@link alignRuns}: that one finds
86
91
  * every run two structures share anywhere (a weave), this one reads two
@@ -195,7 +200,7 @@ export declare function frameSlots(ctx: MindContext, query: Uint8Array, cand: Ui
195
200
  * ANCHOR that the query displaced. Neither implies the other, and the
196
201
  * observed failures pass the restatement guard cleanly.
197
202
  *
198
- * Three conditions, all byte-exact and all necessary:
203
+ * Four conditions, all byte-exact and all necessary:
199
204
  *
200
205
  * 1. the query and the anchor must be ONE STRUCTURE — what they share has to
201
206
  * dominate the query, or the query is not a variant of the anchor at all