@hviana/sema 0.8.1 → 0.8.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (114) hide show
  1. package/AGENTS.md +29 -29
  2. package/TRADEMARKS.md +0 -1
  3. package/dist/src/config.d.ts +28 -0
  4. package/dist/src/config.js +20 -0
  5. package/dist/src/geometry.d.ts +21 -0
  6. package/dist/src/geometry.js +21 -0
  7. package/dist/src/meter.d.ts +76 -0
  8. package/dist/src/meter.js +95 -0
  9. package/dist/src/mind/attention.d.ts +4 -0
  10. package/dist/src/mind/attention.js +165 -16
  11. package/dist/src/mind/canonical.d.ts +16 -0
  12. package/dist/src/mind/canonical.js +41 -0
  13. package/dist/src/mind/corpus.d.ts +40 -0
  14. package/dist/src/mind/corpus.js +149 -0
  15. package/dist/src/mind/graph-search.d.ts +7 -0
  16. package/dist/src/mind/graph-search.js +254 -24
  17. package/dist/src/mind/index.d.ts +3 -1
  18. package/dist/src/mind/index.js +1 -0
  19. package/dist/src/mind/match.d.ts +9 -4
  20. package/dist/src/mind/match.js +147 -61
  21. package/dist/src/mind/mechanisms/cast.js +19 -3
  22. package/dist/src/mind/mechanisms/confluence.js +24 -0
  23. package/dist/src/mind/mechanisms/cover.js +6 -0
  24. package/dist/src/mind/mechanisms/recall.js +32 -4
  25. package/dist/src/mind/mind.d.ts +57 -0
  26. package/dist/src/mind/mind.js +72 -1
  27. package/dist/src/mind/pipeline-mechanism.d.ts +7 -0
  28. package/dist/src/mind/pipeline.js +66 -20
  29. package/dist/src/mind/primitives.js +9 -1
  30. package/dist/src/mind/rationale.d.ts +28 -1
  31. package/dist/src/mind/rationale.js +22 -1
  32. package/dist/src/mind/reasoning.d.ts +25 -3
  33. package/dist/src/mind/reasoning.js +125 -20
  34. package/dist/src/mind/recognition.js +4 -8
  35. package/dist/src/mind/resonance.js +20 -1
  36. package/dist/src/mind/trace.js +1 -0
  37. package/dist/src/mind/traverse.js +15 -3
  38. package/dist/src/mind/types.d.ts +49 -4
  39. package/docs/INVARIANTS.md +2 -2
  40. package/docs/architecture/bounded-reads.md +1 -1
  41. package/docs/architecture/commonality.md +2 -2
  42. package/docs/architecture/cost-model.md +2 -2
  43. package/docs/architecture/determinism.md +7 -7
  44. package/docs/architecture/match-project.md +2 -3
  45. package/docs/architecture/mechanism-market.md +10 -10
  46. package/docs/architecture/meter.md +5 -5
  47. package/docs/architecture/store.md +3 -3
  48. package/docs/failures/tempting-but-wrong.md +34 -6
  49. package/docs/harness/gates.md +2 -2
  50. package/docs/mechanisms/cast.md +2 -2
  51. package/docs/mechanisms/cover.md +2 -3
  52. package/docs/mechanisms/extraction.md +7 -7
  53. package/docs/mechanisms/recall.md +8 -9
  54. package/jsr.json +1 -1
  55. package/package.json +1 -1
  56. package/src/alu/README.md +11 -12
  57. package/src/config.ts +48 -0
  58. package/src/geometry.ts +21 -0
  59. package/src/meter.ts +98 -0
  60. package/src/mind/attention.ts +167 -16
  61. package/src/mind/canonical.ts +43 -0
  62. package/src/mind/corpus.ts +202 -0
  63. package/src/mind/graph-search.ts +277 -23
  64. package/src/mind/index.ts +8 -1
  65. package/src/mind/match.ts +148 -57
  66. package/src/mind/mechanisms/cast.ts +20 -2
  67. package/src/mind/mechanisms/confluence.ts +24 -0
  68. package/src/mind/mechanisms/cover.ts +5 -0
  69. package/src/mind/mechanisms/recall.ts +32 -4
  70. package/src/mind/mind.ts +125 -0
  71. package/src/mind/pipeline-mechanism.ts +7 -0
  72. package/src/mind/pipeline.ts +79 -22
  73. package/src/mind/primitives.ts +9 -1
  74. package/src/mind/rationale.ts +35 -1
  75. package/src/mind/reasoning.ts +145 -13
  76. package/src/mind/recognition.ts +4 -8
  77. package/src/mind/resonance.ts +19 -1
  78. package/src/mind/trace.ts +1 -0
  79. package/src/mind/traverse.ts +16 -6
  80. package/src/mind/types.ts +53 -4
  81. package/test/100-complete-grounding-trace.test.mjs +109 -0
  82. package/test/101-alignment-gap-bound.test.mjs +106 -0
  83. package/test/102-production-composes-at-scale.test.mjs +110 -0
  84. package/test/103-alignment-gap-budget.test.mjs +89 -0
  85. package/test/104-composition-is-reported.test.mjs +90 -0
  86. package/test/105-derive-through-reports-its-refusal.test.mjs +137 -0
  87. package/test/106-the-join-fires.test.mjs +94 -0
  88. package/test/107-the-join-is-counted.test.mjs +81 -0
  89. package/test/108-the-join-chains.test.mjs +78 -0
  90. package/test/109-the-pivot-is-counted.test.mjs +60 -0
  91. package/test/110-the-reasoner-stops-when-the-question-is-answered.test.mjs +91 -0
  92. package/test/111-the-cover-assembly-is-counted.test.mjs +74 -0
  93. package/test/112-the-exploration-does-not-grow-with-the-hub.test.mjs +89 -0
  94. package/test/113-the-rationale-payload-is-bounded.test.mjs +84 -0
  95. package/test/114-alignment-budget-is-per-sweep.test.mjs +93 -0
  96. package/test/116-the-extension-is-gated-by-the-pipelines-own-remainder.test.mjs +100 -0
  97. package/test/117-corpus-search.test.mjs +171 -0
  98. package/test/118-the-join-reaches-a-key-off-the-cut.test.mjs +74 -0
  99. package/test/119-the-work-does-not-grow-with-the-corpus.test.mjs +122 -0
  100. package/test/120-composition-is-consequence.test.mjs +132 -0
  101. package/test/121-the-extension-does-not-grow-with-the-corpus.test.mjs +128 -0
  102. package/test/122-the-climb-search-does-not-grow-with-the-corpus.test.mjs +117 -0
  103. package/test/123-the-paired-formulas-agree.test.mjs +90 -0
  104. package/test/125-the-post-grounding-branch-publishes-its-operand.test.mjs +51 -0
  105. package/test/126-the-pipeline-does-not-name-mechanisms.test.mjs +42 -0
  106. package/test/128-the-leads-somewhere-pair-agrees.test.mjs +83 -0
  107. package/test/129-the-trace-payload-shape.test.mjs +164 -0
  108. package/test/14-scaling.test.mjs +10 -7
  109. package/test/32-confluence.test.mjs +68 -0
  110. package/test/38-reason-restate-guard.test.mjs +8 -2
  111. package/test/43-cast-analog-seat.test.mjs +10 -0
  112. package/test/55-cost-meter.test.mjs +859 -0
  113. package/test/76-reference-binding.test.mjs +6 -1
  114. package/test/89-completion-recursion.test.mjs +30 -5
@@ -238,6 +238,14 @@ export interface DerivationItem {
238
238
  * {@link GraphSearch}'s rules fired, recovered from the rule's premise/
239
239
  * conclusion shape (the rules carry no label, so this classifies by structure,
240
240
  * the single place that maps rule geometry to a name). */
241
+ // CLOSED ON PURPOSE — AND ONLY THIS ONE IS. A derivation move is BRANCHED ON
242
+ // (`classifyMove`, the rationale's readers, MOVE_NOTE's fallback), so it is a
243
+ // closed union: adding one without teaching every reader is a compile error,
244
+ // which is what a closed vocabulary buys. The MECHANISM names in
245
+ // `rationale.ts` are the opposite case — written and displayed, never
246
+ // branched on — and they stay free strings that COMPOSE with the nesting
247
+ // (`["respond", "think", "recognise"]`). That asymmetry is deliberate; do not
248
+ // "fix" it by uniting the two (see test/126 for the pipeline half of it).
241
249
  export type DerivationMove =
242
250
  | "axiom" // a seed: a perceived leaf, a recognised form, or a computed result
243
251
  | "follow-edge" // form→form via a continuation edge (STEP) — the core "what follows what"
@@ -375,6 +383,14 @@ export class GraphSearch {
375
383
  private readonly host: GraphSearchHost,
376
384
  ) {}
377
385
 
386
+ /** The nodes the QUERY canonically names — the same identity the store's keys
387
+ * were written through. A byte-exact test is not enough: the query writes
388
+ * `Eiffel Tower country` and the deposited node is `eiffel tower country`, so
389
+ * a join that filters the query's own subject by RAW bytes re-admits it —
390
+ * measured: that is the trap's wrong answer (`The capital of Eiffel Tower
391
+ * country is Berlin.`). Cached by query identity, because the search is
392
+ * reused across responses. */
393
+
378
394
  /* * The hub bound √N (bounded-reads.md) — the ONE
379
395
  * fan-out cap, stated here rather than imported from `traverse.ts` because
380
396
  * this module is deliberately host-based (it holds a bare Store, never a
@@ -703,7 +719,13 @@ export class GraphSearch {
703
719
  return this.coverRules(it, coversDone, coverableByStart);
704
720
  }
705
721
  if (it.kind === "form") {
706
- return this.formRules(it, conceptTarget, substitutions, nodeBytes);
722
+ return this.formRules(
723
+ it,
724
+ conceptTarget,
725
+ substitutions,
726
+ nodeBytes,
727
+ queryLen,
728
+ );
707
729
  }
708
730
  return this.outRules(it, {
709
731
  W,
@@ -808,6 +830,7 @@ export class GraphSearch {
808
830
  conceptTarget: ReadonlyMap<number, number>,
809
831
  substitutions: ReadonlyMap<number, Uint8Array> | undefined,
810
832
  nodeBytes: (n: number) => Uint8Array,
833
+ queryLen: number,
811
834
  ): Iterable<Rule<GItem>> {
812
835
  // Articulation: emit voice bytes at the recognised span; the hop/concept/
813
836
  // emit chain is suppressed — the form contributes only its substitute.
@@ -843,7 +866,45 @@ export class GraphSearch {
843
866
  // guard then dead-ends it) with no way to reach the forward edge.
844
867
  // Forking offers every continuation as its own rule so the one that
845
868
  // genuinely advances (not a duplicate) is still reachable.
869
+ // A CHAIN HOP OFFERS ONLY WHAT THE QUESTION CAN PAY FOR.
870
+ //
871
+ // `hubBound` = √N is the READ cap — every read here stays inside it — but
872
+ // it is not an exploration bound: measured, a hub of degree 1083 sits
873
+ // BELOW √N = 1559, so a hop offered all 1083 continuations, the chart grew
874
+ // to 3113 outs for a two-word question, and since every out with an
875
+ // uncovered tail probes its tail's prefixes (measured: 16 885 canonical
876
+ // probes = 87% of that query's work, and its 270 MB peak / 256 MB OOM),
877
+ // the cost came from OFFERING rather than from reading.
878
+ //
879
+ // The bound is derived, not tuned: a derivation of L hops consumes ~L
880
+ // units of the question, so a hop cannot be paid for by offering more
881
+ // continuations than the question has units —
882
+ // `ceil(queryLen / W)`, floored at 2 for plurality. It is QUERY-sized
883
+ // (invariant 5: no per-query read grows with N) and it leaves `hubBound`
884
+ // and every read untouched.
885
+ // THE OFFER IS THE CORPUS'S OWN STRUCTURE, and the search pays for
886
+ // exploring it. There is no offer cap here any more: the traversal cap I
887
+ // had put on this hop was a short-circuit — it bounded what a hop could
888
+ // OFFER instead of charging for it — and it was not needed.
889
+ //
890
+ // MEASURED in the regime where it used to bite (`hubBound = ceil(√N)`
891
+ // GREATER than the hub's degree — reached in a fixture by choosing the
892
+ // degree below √N, so the trained store is not needed): with the cap the
893
+ // offer was 8/9/9 continuations at degrees 35/70/120; without it, 52/84/120
894
+ // — and the WORK is LINEAR in the degree, not quadratic: pushes 262/296/332,
895
+ // perceptions 530/592/757, while the PEAK is identical with and without the
896
+ // cap (218/415/689 MB against 215/410/662) because it is set by the store,
897
+ // not by the fan-out. What made this hop expensive was never the fan-out
898
+ // breadth: it was the per-offer work, two duplicate/oversized computations
899
+ // since removed (the per-offset canonical scan, and the tail scan now
900
+ // restricted to the fold's boundaries).
901
+ //
902
+ // The residual, stated: the trained store's hub (degree 1 083) is an
903
+ // EXTRAPOLATION from this linear shape, not a measurement.
846
904
  const nx = this.store.nextFirst(it.node, this.hubBound());
905
+ // Count what is OFFERED, not what was read: the evidence-preferred
906
+ // continuation is yielded too, even when it lies outside the cap.
907
+ if (this.host.meter) this.host.meter.chainOffers += nx.length + 1;
847
908
  if (nx.length) {
848
909
  // The SAME evidence-weighted disambiguation the first hop uses
849
910
  // (below) identifies the most-corroborated continuation. Yielding
@@ -852,10 +913,14 @@ export class GraphSearch {
852
913
  // arrivals at an EQUAL cost (`cost < current`, strictly) — so
853
914
  // among same-depth sibling forks that tie in cost, the
854
915
  // evidence-backed edge wins deterministically, never by
855
- // exploration-order luck. `preferred`, when set, is necessarily
856
- // an element of `nx` (chooseNext reads the identical hub-bounded
857
- // set — see traverse.ts), so a plain skip-in-place suffices; no
858
- // second array need be allocated to reorder it to the front.
916
+ // exploration-order luck. `preferred` is NOT necessarily an element
917
+ // of `nx` any more: `nx` is the capped read above, while `chooseNext`
918
+ // reads its own hub-bounded set (see traverse.ts) — so the evidence
919
+ // pick may lie outside the cap, and it is still yielded FIRST on
920
+ // purpose. The cap bounds what the hop EXPLORES; it must never make
921
+ // the evidence-ranked continuation unreachable, or the derivation the
922
+ // corpus corroborates would lose to exploration order. Nothing is
923
+ // reallocated: `nx` is walked with a skip-in-place for the duplicate.
859
924
  const preferred = nx.length > 1
860
925
  ? this.host.chooseNext?.(it.node)
861
926
  : undefined;
@@ -1078,12 +1143,51 @@ export class GraphSearch {
1078
1143
  // concepts/connectors either (those need the caller's async
1079
1144
  // pre-resolution) — the recursion follows edges and fusion, which is what
1080
1145
  // a deeper rewrite chain is made of.
1146
+ if (this.host.meter) this.host.meter.recompletes++;
1081
1147
  const rec = this.host.recogniseSpan(bytes);
1082
1148
  const kids = new Set(nrec.kids);
1149
+ // THE NODE'S OWN KIDS ARE SITES BY STRUCTURE — recognition cannot be the
1150
+ // only source. A produced composite is decomposed by ITS OWN SHAPE, and
1151
+ // at hub scale the recognition of a produced span returns the WHOLE while
1152
+ // deliberately suppressing its atoms (the off-boundary suppression), so
1153
+ // the kid filter below would admit nothing at all and the chain would end
1154
+ // at the intermediate composite. Measured on the chain
1155
+ // `seed → "p q" → (p→r, q→s) → "r s" → "m n"`: below the flip it reaches
1156
+ // "m n" with fuse+recompose, above it stops at "p q" — the trace shows
1157
+ // `recognise("p q") ⇒ form "p q"` alone, no parts.
1158
+ //
1159
+ // Laying the node's kids out over its own bytes restores exactly the
1160
+ // decomposition the node's tree already states; a kid recognition ALREADY
1161
+ // offers is skipped, so below the flip the seed set is byte-identical to
1162
+ // what it was. The filter's guarantee is untouched: nothing beyond the
1163
+ // node's own kids may enter.
1164
+ const recognised = rec.sites.filter((s) => kids.has(s.payload));
1165
+ // DEDUPED BY SPAN, not by payload. A node whose kids repeat (`"abab"`
1166
+ // folds as ["ab","ab"]) has TWO occurrences of the same node at different
1167
+ // offsets; a payload-keyed set suppressed the structural site for BOTH, so
1168
+ // the second occurrence had no site at all. The set now holds the spans
1169
+ // recognition already covers, and a structural site is added exactly when
1170
+ // nothing covers that PLACE — O(1) lookups, no extra reads.
1171
+ const seenSpans = new Set(recognised.map((s) => `${s.start}:${s.end}`));
1172
+ const structural: Site[] = [];
1173
+ {
1174
+ let off = 0;
1175
+ for (const k of nrec.kids) {
1176
+ const len = this.store.bytesPrefix(k, ALL).length;
1177
+ if (!seenSpans.has(`${off}:${off + len}`) && len > 0) {
1178
+ structural.push({
1179
+ start: off,
1180
+ end: Math.min(off + len, bytes.length),
1181
+ payload: k,
1182
+ });
1183
+ }
1184
+ off += len;
1185
+ }
1186
+ }
1083
1187
  const solved = this.solve(
1084
1188
  bytes.length,
1085
1189
  {
1086
- sites: rec.sites.filter((s) => kids.has(s.payload)),
1190
+ sites: [...recognised, ...structural],
1087
1191
  leaves: rec.leaves,
1088
1192
  splits: rec.splits,
1089
1193
  starts: rec.starts,
@@ -1163,6 +1267,14 @@ export class GraphSearch {
1163
1267
  if (!this.host.recogniseSpan) return;
1164
1268
  const tail = queryBytes.subarray(fact.j, queryLen);
1165
1269
  if (tail.length === 0) return;
1270
+ // Report ONLY the invocations that could have joined: the search asks this
1271
+ // rule for every finalized out with a node, which includes the one-byte
1272
+ // outs the cover bridges with — measured, 68 refusals for a single
1273
+ // 3-relation query, all of them letters. A form shorter than one window is
1274
+ // not a fact a join could travel through, so it is not a refusal worth
1275
+ // reporting; W is the same line the rest of the mind draws between a chance
1276
+ // overlap and a form.
1277
+ const reportable = fact.bytes.length >= this.maxGroup;
1166
1278
  // The entity candidates are the forms the fact's own bytes CONTAIN — the
1167
1279
  // same recogniser the query went through, so the evidence standard is the
1168
1280
  // query's. A byte atom is never a subject; the fact's own node is the span
@@ -1171,12 +1283,66 @@ export class GraphSearch {
1171
1283
  // LENDS it when it can (Mind does, with the response-scoped struct cache),
1172
1284
  // and a bare host falls back to the raw-store probe, so the search stays
1173
1285
  // host-based.
1174
- const leading = this.host.recogniseSpan(fact.bytes).sites.filter((s) =>
1175
- s.payload >= 0 && s.payload !== fact.node &&
1176
- (this.host.leadsSomewhere !== undefined
1177
- ? this.host.leadsSomewhere(s.payload)
1178
- : this.store.hasNext(s.payload) || this.store.hasHalo(s.payload))
1286
+ const factRec = this.host.recogniseSpan(fact.bytes);
1287
+ const leads = (id: number): boolean =>
1288
+ this.host.leadsSomewhere !== undefined
1289
+ ? this.host.leadsSomewhere(id)
1290
+ : this.store.hasNext(id) || this.store.hasHalo(id);
1291
+ // THE QUERY'S OWN SUBJECT, CANONICALLY. The filter used raw bytes and the
1292
+ // store's nodes are canonical, so `Eiffel Tower country` in the query did
1293
+ // not match the deposited `eiffel tower country` — measured, that is the
1294
+ // trap's wrong answer.
1295
+ //
1296
+ // TAKEN FROM THE RECOGNITION THE RESPONSE ALREADY COMPUTED — the host's
1297
+ // `recogniseSpan`, the same surface the rest of this search uses — not from
1298
+ // a second offset scan. `canonicalQueryNodes` re-derived, per byte offset,
1299
+ // what `recognise` had already resolved once per query (its memo is keyed by
1300
+ // content), and that scan was the largest single cost the DIANOT join added:
1301
+ // measured against the pre-change tree, the same fixture and the same test
1302
+ // were 21 s slower with the scan than without it. A recognised site IS a
1303
+ // canonical node of the query that can lead somewhere, which is exactly the
1304
+ // set this filter wants, and it costs nothing to read.
1305
+ const queryNodes = new Set<number>(
1306
+ (this.host.recogniseSpan?.(queryBytes)?.sites ?? []).map((s) =>
1307
+ s.payload
1308
+ ),
1179
1309
  );
1310
+ // TWO SOURCES, ONE ADMISSION. The recognition of a STORED WHOLE returns the
1311
+ // whole and stops — measured: for `The director of Eva is Gustaf Molander.`
1312
+ // it yields exactly ONE site, the fact's own node — so the entity a join
1313
+ // exists for is never proposed. The canonical fold is the second source,
1314
+ // and the scan runs only for a FORM (≥ W: a one-byte out is not something to
1315
+ // join through, and running it per letter measured 20-26 s in test/99).
1316
+ const W = this.maxGroup;
1317
+ const proposed = new Map<number, Uint8Array>();
1318
+ // The SOURCE of each proposal travels with it: a refusal that names only the
1319
+ // bytes leaves the next reader guessing which path proposed them — three
1320
+ // attempts at the chained join were spent fixing paths that never produced
1321
+ // the offending candidate.
1322
+ const source = new Map<number, string>();
1323
+ for (const s of factRec.sites) {
1324
+ if (s.payload >= 0 && leads(s.payload)) {
1325
+ proposed.set(s.payload, this.store.bytesPrefix(s.payload, ALL));
1326
+ source.set(s.payload, "recognised site");
1327
+ }
1328
+ }
1329
+ if (this.host.canonResolve !== undefined && fact.bytes.length >= W) {
1330
+ const canon = this.host.canonResolve.bind(this.host);
1331
+ for (let start = 0; start < fact.bytes.length; start++) {
1332
+ for (let end = fact.bytes.length; end - start >= W; end--) {
1333
+ const id = canon(fact.bytes.subarray(start, end));
1334
+ if (id === null) continue;
1335
+ if (leads(id)) {
1336
+ proposed.set(id, this.store.bytesPrefix(id, ALL));
1337
+ source.set(id, "canonical fold");
1338
+ }
1339
+ break; // the longest form at this offset wins
1340
+ }
1341
+ }
1342
+ }
1343
+ const leading = [...proposed]
1344
+ .filter(([payload]) => payload !== fact.node && !queryNodes.has(payload))
1345
+ .map(([payload, bytes]) => ({ payload, bytes }));
1180
1346
  // …then prefer the entity the query did NOT name, and the MAXIMAL one. The
1181
1347
  // join exists to reach the subject the query never wrote, so:
1182
1348
  // • a candidate the query already contains is the query's OWN subject, and
@@ -1187,32 +1353,120 @@ export class GraphSearch {
1187
1353
  // introduces ("Timur" must not win over "Timur Bekmambetov").
1188
1354
  // Byte work over bytes already read, and the pruning REMOVES the
1189
1355
  // resolve()/nextFirst() probes these candidates would have paid.
1356
+ if (leading.length === 0) {
1357
+ if (this.host.meter) this.host.meter.joinNoEntity++;
1358
+ if (reportable) {
1359
+ // Report WHAT the recognition returned, not just that nothing led: the
1360
+ // count and the first few site texts are the difference between "the
1361
+ // fact was not recognised" and "it was recognised but nothing led".
1362
+ const seen = factRec.sites.slice(0, 3).map((s) =>
1363
+ this.store.bytesPrefix(s.payload, ALL)
1364
+ );
1365
+ this.host.reportSearch?.(
1366
+ "deriveThroughMiss",
1367
+ [fact.bytes, tail, ...seen],
1368
+ `no entity inside the fact leads anywhere — ${factRec.sites.length} site(s) recognised inside it`,
1369
+ );
1370
+ }
1371
+ }
1190
1372
  const candidates = leading
1191
- .map((s) => ({
1192
- payload: s.payload,
1193
- bytes: this.store.bytesPrefix(s.payload, ALL),
1194
- }))
1195
- .filter((c) => indexOf(queryBytes, c.bytes, 0) < 0)
1196
1373
  .filter((c, _i, all) =>
1197
1374
  !all.some((o) =>
1198
1375
  o.bytes.length > c.bytes.length && indexOf(o.bytes, c.bytes, 0) >= 0
1199
1376
  )
1200
1377
  );
1201
1378
  for (const c of candidates) {
1202
- const key = this.host.resolve(concat2(c.bytes, tail));
1203
- if (key === null) continue;
1204
- const nx = this.store.nextFirst(key, 1);
1205
- if (nx.length === 0) continue;
1379
+ // THE SHORTEST TAIL PREFIX WHOSE KEY ALSO LEADS SOMEWHERE.
1380
+ //
1381
+ // The conclusion covers only that prefix, so the rest of the tail stays
1382
+ // for the step after — which is what CHAINING is (a whole-tail key
1383
+ // consumed the whole remainder and made every join terminal: measured,
1384
+ // `joinFired=0` on a three-relation query).
1385
+ //
1386
+ // A KEY THAT RESOLVES IS NOT ENOUGH. Measured with a dry run of these
1387
+ // very primitives: for the candidate `Sweden` the first tail prefix that
1388
+ // resolves is `" "` — the key `Sweden ` (a trailing space) — and it leads
1389
+ // NOWHERE (`nextFirst` = 0), so accepting it refused the join while the
1390
+ // key that names the fact (`sweden capital`) sat one prefix further. The
1391
+ // loop therefore asks BOTH questions before accepting, and keeps looking
1392
+ // otherwise.
1393
+ //
1394
+ // EXACT FIRST, THEN CANONICAL: the corpus holds both identities.
1395
+ let key: number | null = null;
1396
+ let used = 0;
1397
+ // The continuation the accepted key leads to, carried out of the loop:
1398
+ // the loop already HAD to read it to accept the key (a key that leads
1399
+ // nowhere is not the relation), so re-reading it after the loop was a
1400
+ // duplicate read and a branch that could never be taken.
1401
+ let next: number | null = null;
1402
+ let keyBytes = c.bytes;
1403
+ // THE CANDIDATE ENDS ARE THE PREFIXES THAT ARE STORED NODES, ASCENDING.
1404
+ // The key is `entity + prefix`, and it names a relation exactly when that
1405
+ // concatenation IS a node — so the ends come from a content-addressed
1406
+ // probe per offset (the host's `contentKeyEnds`, the learning path's own
1407
+ // mechanism: one leaf walk plus one `findBranch` per offset, no `resolve`),
1408
+ // never from the fold's boundaries. A boundary is not a proxy: a stored
1409
+ // member's end is the end of ITS OWN stream, and the fold never cuts at a
1410
+ // stream's end — measured, "stockholm mayor" exists, leads on to the mayor
1411
+ // fact, and its boundary 6 sits in neither the tail's cuts ([4,7]) nor the
1412
+ // concatenation's. Filtering the scan by "is this a node?" cannot change
1413
+ // the winner: a position that is not a node cannot resolve, so skipping it
1414
+ // is invisible; and the order stays SHORTEST FIRST, which is a semantic
1415
+ // law, not an optimisation (test/106, test/108 pin it).
1416
+ //
1417
+ // A host that cannot answer falls back to every prefix: exact and
1418
+ // complete, at a `resolve` per offset. A host that CAN answer is
1419
+ // authoritative even when it answers "none" — if no prefix is a node then
1420
+ // no key exists to resolve, so enumerating would only pay nulls. (A key
1421
+ // reachable through the CANONICAL equivalence alone and ending off every
1422
+ // node end is therefore not tried here; that dimension is unreachable on
1423
+ // this path by construction and is not part of the exact-key law.)
1424
+ const ends = this.host.contentKeyEnds?.(c.bytes, tail);
1425
+ const candidateEnds = function* (): Generator<number> {
1426
+ if (ends !== undefined) {
1427
+ yield* ends;
1428
+ return;
1429
+ }
1430
+ for (let p = 1; p <= tail.length; p++) yield p;
1431
+ };
1432
+ for (const len of candidateEnds()) {
1433
+ keyBytes = concat2(c.bytes, tail.subarray(0, len));
1434
+ const k = this.host.resolve(keyBytes) ??
1435
+ this.host.canonResolve?.(keyBytes) ??
1436
+ null;
1437
+ if (k === null) continue;
1438
+ const nx = this.store.nextFirst(k, 1);
1439
+ if (nx.length === 0) continue;
1440
+ key = k;
1441
+ used = len;
1442
+ next = nx[0];
1443
+ break;
1444
+ }
1445
+ if (key === null) {
1446
+ if (this.host.meter) this.host.meter.joinNoKey++;
1447
+ if (reportable) {
1448
+ this.host.reportSearch?.(
1449
+ "deriveThroughMiss",
1450
+ [c.bytes, tail, keyBytes],
1451
+ `no learnt key names this entity and tail together ` +
1452
+ `(candidate #${c.payload}, from the ${
1453
+ source.get(c.payload) ?? "unknown"
1454
+ } source)`,
1455
+ );
1456
+ }
1457
+ continue;
1458
+ }
1459
+ if (this.host.meter) this.host.meter.joinFired++;
1206
1460
  yield {
1207
1461
  premises: [fact],
1208
1462
  conclusion: {
1209
1463
  kind: "out",
1210
1464
  i: fact.i,
1211
- j: queryLen,
1212
- bytes: this.store.bytesPrefix(nx[0], ALL),
1465
+ j: fact.j + used,
1466
+ bytes: this.store.bytesPrefix(next!, ALL),
1213
1467
  cover: true,
1214
1468
  rec: true,
1215
- node: nx[0],
1469
+ node: next!,
1216
1470
  throughFact: true,
1217
1471
  },
1218
1472
  cost: STEP,
package/src/mind/index.ts CHANGED
@@ -4,7 +4,12 @@
4
4
  // exported from mind/mind.ts directly.
5
5
 
6
6
  export { Mind } from "./mind.js";
7
- export type { Input, Response } from "./mind.js";
7
+ export type {
8
+ CorpusTextPair,
9
+ CorpusTextResult,
10
+ Input,
11
+ Response,
12
+ } from "./mind.js";
8
13
  export type { ComputedSpan, ExtensionHost } from "./mind.js";
9
14
  export type {
10
15
  MechanismResult,
@@ -38,3 +43,5 @@ export type {
38
43
  NarrowDecisionData,
39
44
  Provenance,
40
45
  } from "./pipeline.js";
46
+ export { sampleCorpus, searchCorpus } from "./corpus.js";
47
+ export type { CorpusMiss, CorpusPair, CorpusResult } from "./corpus.js";