@hviana/sema 0.8.1 → 0.8.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (57) hide show
  1. package/dist/src/config.d.ts +17 -0
  2. package/dist/src/config.js +18 -0
  3. package/dist/src/meter.d.ts +25 -0
  4. package/dist/src/meter.js +44 -0
  5. package/dist/src/mind/corpus.d.ts +40 -0
  6. package/dist/src/mind/corpus.js +149 -0
  7. package/dist/src/mind/graph-search.d.ts +7 -0
  8. package/dist/src/mind/graph-search.js +235 -24
  9. package/dist/src/mind/index.d.ts +3 -1
  10. package/dist/src/mind/index.js +1 -0
  11. package/dist/src/mind/match.d.ts +8 -3
  12. package/dist/src/mind/match.js +142 -58
  13. package/dist/src/mind/mechanisms/cast.js +18 -2
  14. package/dist/src/mind/mechanisms/cover.js +6 -0
  15. package/dist/src/mind/mind.d.ts +55 -0
  16. package/dist/src/mind/mind.js +72 -2
  17. package/dist/src/mind/pipeline.js +25 -6
  18. package/dist/src/mind/reasoning.d.ts +5 -1
  19. package/dist/src/mind/reasoning.js +54 -1
  20. package/dist/src/mind/traverse.js +9 -1
  21. package/dist/src/mind/types.d.ts +22 -0
  22. package/docs/failures/tempting-but-wrong.md +31 -2
  23. package/jsr.json +1 -1
  24. package/package.json +1 -1
  25. package/src/config.ts +35 -0
  26. package/src/meter.ts +47 -0
  27. package/src/mind/corpus.ts +202 -0
  28. package/src/mind/graph-search.ts +252 -23
  29. package/src/mind/index.ts +8 -1
  30. package/src/mind/match.ts +143 -54
  31. package/src/mind/mechanisms/cast.ts +17 -1
  32. package/src/mind/mechanisms/cover.ts +5 -0
  33. package/src/mind/mind.ts +123 -0
  34. package/src/mind/pipeline.ts +30 -6
  35. package/src/mind/reasoning.ts +55 -0
  36. package/src/mind/traverse.ts +9 -1
  37. package/src/mind/types.ts +26 -0
  38. package/test/100-complete-grounding-trace.test.mjs +109 -0
  39. package/test/101-alignment-gap-bound.test.mjs +106 -0
  40. package/test/102-production-composes-at-scale.test.mjs +110 -0
  41. package/test/103-alignment-gap-budget.test.mjs +89 -0
  42. package/test/104-composition-is-reported.test.mjs +90 -0
  43. package/test/105-derive-through-reports-its-refusal.test.mjs +113 -0
  44. package/test/106-the-join-fires.test.mjs +94 -0
  45. package/test/107-the-join-is-counted.test.mjs +81 -0
  46. package/test/108-the-join-chains.test.mjs +78 -0
  47. package/test/109-the-pivot-is-counted.test.mjs +60 -0
  48. package/test/110-the-reasoner-stops-when-the-question-is-answered.test.mjs +91 -0
  49. package/test/111-the-cover-assembly-is-counted.test.mjs +74 -0
  50. package/test/112-the-exploration-does-not-grow-with-the-hub.test.mjs +89 -0
  51. package/test/113-the-rationale-payload-is-bounded.test.mjs +84 -0
  52. package/test/114-alignment-budget-is-per-sweep.test.mjs +93 -0
  53. package/test/116-the-extension-is-gated-by-the-pipelines-own-remainder.test.mjs +100 -0
  54. package/test/117-corpus-search.test.mjs +171 -0
  55. package/test/14-scaling.test.mjs +10 -7
  56. package/test/76-reference-binding.test.mjs +6 -1
  57. package/test/89-completion-recursion.test.mjs +30 -5
@@ -194,6 +194,13 @@ export class GraphSearch {
194
194
  this.maxGroup = maxGroup;
195
195
  this.host = host;
196
196
  }
197
+ /** The nodes the QUERY canonically names — the same identity the store's keys
198
+ * were written through. A byte-exact test is not enough: the query writes
199
+ * `Eiffel Tower country` and the deposited node is `eiffel tower country`, so
200
+ * a join that filters the query's own subject by RAW bytes re-admits it —
201
+ * measured: that is the trap's wrong answer (`The capital of Eiffel Tower
202
+ * country is Berlin.`). Cached by query identity, because the search is
203
+ * reused across responses. */
197
204
  /* * The hub bound √N (bounded-reads.md) — the ONE
198
205
  * fan-out cap, stated here rather than imported from `traverse.ts` because
199
206
  * this module is deliberately host-based (it holds a bare Store, never a
@@ -461,7 +468,7 @@ export class GraphSearch {
461
468
  return this.coverRules(it, coversDone, coverableByStart);
462
469
  }
463
470
  if (it.kind === "form") {
464
- return this.formRules(it, conceptTarget, substitutions, nodeBytes);
471
+ return this.formRules(it, conceptTarget, substitutions, nodeBytes, queryLen);
465
472
  }
466
473
  return this.outRules(it, {
467
474
  W,
@@ -547,7 +554,7 @@ export class GraphSearch {
547
554
  }
548
555
  /** form(i,j,node,via): follow the graph out of `node`, or (in articulation)
549
556
  * emit its substitute voice directly. */
550
- *formRules(it, conceptTarget, substitutions, nodeBytes) {
557
+ *formRules(it, conceptTarget, substitutions, nodeBytes, queryLen) {
551
558
  // Articulation: emit voice bytes at the recognised span; the hop/concept/
552
559
  // emit chain is suppressed — the form contributes only its substitute.
553
560
  if (substitutions) {
@@ -581,7 +588,46 @@ export class GraphSearch {
581
588
  // guard then dead-ends it) with no way to reach the forward edge.
582
589
  // Forking offers every continuation as its own rule so the one that
583
590
  // genuinely advances (not a duplicate) is still reachable.
591
+ // A CHAIN HOP OFFERS ONLY WHAT THE QUESTION CAN PAY FOR.
592
+ //
593
+ // `hubBound` = √N is the READ cap — every read here stays inside it — but
594
+ // it is not an exploration bound: measured, a hub of degree 1083 sits
595
+ // BELOW √N = 1559, so a hop offered all 1083 continuations, the chart grew
596
+ // to 3113 outs for a two-word question, and since every out with an
597
+ // uncovered tail probes its tail's prefixes (measured: 16 885 canonical
598
+ // probes = 87% of that query's work, and its 270 MB peak / 256 MB OOM),
599
+ // the cost came from OFFERING rather than from reading.
600
+ //
601
+ // The bound is derived, not tuned: a derivation of L hops consumes ~L
602
+ // units of the question, so a hop cannot be paid for by offering more
603
+ // continuations than the question has units —
604
+ // `ceil(queryLen / W)`, floored at 2 for plurality. It is QUERY-sized
605
+ // (invariant 5: no per-query read grows with N) and it leaves `hubBound`
606
+ // and every read untouched.
607
+ // THE OFFER IS THE CORPUS'S OWN STRUCTURE, and the search pays for
608
+ // exploring it. There is no offer cap here any more: the traversal cap I
609
+ // had put on this hop was a short-circuit — it bounded what a hop could
610
+ // OFFER instead of charging for it — and it was not needed.
611
+ //
612
+ // MEASURED in the regime where it used to bite (`hubBound = ceil(√N)`
613
+ // GREATER than the hub's degree — reached in a fixture by choosing the
614
+ // degree below √N, so the trained store is not needed): with the cap the
615
+ // offer was 8/9/9 continuations at degrees 35/70/120; without it, 52/84/120
616
+ // — and the WORK is LINEAR in the degree, not quadratic: pushes 262/296/332,
617
+ // perceptions 530/592/757, while the PEAK is identical with and without the
618
+ // cap (218/415/689 MB against 215/410/662) because it is set by the store,
619
+ // not by the fan-out. What made this hop expensive was never the fan-out
620
+ // breadth: it was the per-offer work, two duplicate/oversized computations
621
+ // since removed (the per-offset canonical scan, and the tail scan now
622
+ // restricted to the fold's boundaries).
623
+ //
624
+ // The residual, stated: the trained store's hub (degree 1 083) is an
625
+ // EXTRAPOLATION from this linear shape, not a measurement.
584
626
  const nx = this.store.nextFirst(it.node, this.hubBound());
627
+ // Count what is OFFERED, not what was read: the evidence-preferred
628
+ // continuation is yielded too, even when it lies outside the cap.
629
+ if (this.host.meter)
630
+ this.host.meter.chainOffers += nx.length + 1;
585
631
  if (nx.length) {
586
632
  // The SAME evidence-weighted disambiguation the first hop uses
587
633
  // (below) identifies the most-corroborated continuation. Yielding
@@ -590,10 +636,14 @@ export class GraphSearch {
590
636
  // arrivals at an EQUAL cost (`cost < current`, strictly) — so
591
637
  // among same-depth sibling forks that tie in cost, the
592
638
  // evidence-backed edge wins deterministically, never by
593
- // exploration-order luck. `preferred`, when set, is necessarily
594
- // an element of `nx` (chooseNext reads the identical hub-bounded
595
- // set — see traverse.ts), so a plain skip-in-place suffices; no
596
- // second array need be allocated to reorder it to the front.
639
+ // exploration-order luck. `preferred` is NOT necessarily an element
640
+ // of `nx` any more: `nx` is the capped read above, while `chooseNext`
641
+ // reads its own hub-bounded set (see traverse.ts) — so the evidence
642
+ // pick may lie outside the cap, and it is still yielded FIRST on
643
+ // purpose. The cap bounds what the hop EXPLORES; it must never make
644
+ // the evidence-ranked continuation unreachable, or the derivation the
645
+ // corpus corroborates would lose to exploration order. Nothing is
646
+ // reallocated: `nx` is walked with a skip-in-place for the duplicate.
597
647
  const preferred = nx.length > 1
598
648
  ? this.host.chooseNext?.(it.node)
599
649
  : undefined;
@@ -820,8 +870,46 @@ export class GraphSearch {
820
870
  // a deeper rewrite chain is made of.
821
871
  const rec = this.host.recogniseSpan(bytes);
822
872
  const kids = new Set(nrec.kids);
873
+ // THE NODE'S OWN KIDS ARE SITES BY STRUCTURE — recognition cannot be the
874
+ // only source. A produced composite is decomposed by ITS OWN SHAPE, and
875
+ // at hub scale the recognition of a produced span returns the WHOLE while
876
+ // deliberately suppressing its atoms (the off-boundary suppression), so
877
+ // the kid filter below would admit nothing at all and the chain would end
878
+ // at the intermediate composite. Measured on the chain
879
+ // `seed → "p q" → (p→r, q→s) → "r s" → "m n"`: below the flip it reaches
880
+ // "m n" with fuse+recompose, above it stops at "p q" — the trace shows
881
+ // `recognise("p q") ⇒ form "p q"` alone, no parts.
882
+ //
883
+ // Laying the node's kids out over its own bytes restores exactly the
884
+ // decomposition the node's tree already states; a kid recognition ALREADY
885
+ // offers is skipped, so below the flip the seed set is byte-identical to
886
+ // what it was. The filter's guarantee is untouched: nothing beyond the
887
+ // node's own kids may enter.
888
+ const recognised = rec.sites.filter((s) => kids.has(s.payload));
889
+ // DEDUPED BY SPAN, not by payload. A node whose kids repeat (`"abab"`
890
+ // folds as ["ab","ab"]) has TWO occurrences of the same node at different
891
+ // offsets; a payload-keyed set suppressed the structural site for BOTH, so
892
+ // the second occurrence had no site at all. The set now holds the spans
893
+ // recognition already covers, and a structural site is added exactly when
894
+ // nothing covers that PLACE — O(1) lookups, no extra reads.
895
+ const seenSpans = new Set(recognised.map((s) => `${s.start}:${s.end}`));
896
+ const structural = [];
897
+ {
898
+ let off = 0;
899
+ for (const k of nrec.kids) {
900
+ const len = this.store.bytesPrefix(k, ALL).length;
901
+ if (!seenSpans.has(`${off}:${off + len}`) && len > 0) {
902
+ structural.push({
903
+ start: off,
904
+ end: Math.min(off + len, bytes.length),
905
+ payload: k,
906
+ });
907
+ }
908
+ off += len;
909
+ }
910
+ }
823
911
  const solved = this.solve(bytes.length, {
824
- sites: rec.sites.filter((s) => kids.has(s.payload)),
912
+ sites: [...recognised, ...structural],
825
913
  leaves: rec.leaves,
826
914
  splits: rec.splits,
827
915
  starts: rec.starts,
@@ -892,6 +980,14 @@ export class GraphSearch {
892
980
  const tail = queryBytes.subarray(fact.j, queryLen);
893
981
  if (tail.length === 0)
894
982
  return;
983
+ // Report ONLY the invocations that could have joined: the search asks this
984
+ // rule for every finalized out with a node, which includes the one-byte
985
+ // outs the cover bridges with — measured, 68 refusals for a single
986
+ // 3-relation query, all of them letters. A form shorter than one window is
987
+ // not a fact a join could travel through, so it is not a refusal worth
988
+ // reporting; W is the same line the rest of the mind draws between a chance
989
+ // overlap and a form.
990
+ const reportable = fact.bytes.length >= this.maxGroup;
895
991
  // The entity candidates are the forms the fact's own bytes CONTAIN — the
896
992
  // same recogniser the query went through, so the evidence standard is the
897
993
  // query's. A byte atom is never a subject; the fact's own node is the span
@@ -900,10 +996,62 @@ export class GraphSearch {
900
996
  // LENDS it when it can (Mind does, with the response-scoped struct cache),
901
997
  // and a bare host falls back to the raw-store probe, so the search stays
902
998
  // host-based.
903
- const leading = this.host.recogniseSpan(fact.bytes).sites.filter((s) => s.payload >= 0 && s.payload !== fact.node &&
904
- (this.host.leadsSomewhere !== undefined
905
- ? this.host.leadsSomewhere(s.payload)
906
- : this.store.hasNext(s.payload) || this.store.hasHalo(s.payload)));
999
+ const factRec = this.host.recogniseSpan(fact.bytes);
1000
+ const leads = (id) => this.host.leadsSomewhere !== undefined
1001
+ ? this.host.leadsSomewhere(id)
1002
+ : this.store.hasNext(id) || this.store.hasHalo(id);
1003
+ // THE QUERY'S OWN SUBJECT, CANONICALLY. The filter used raw bytes and the
1004
+ // store's nodes are canonical, so `Eiffel Tower country` in the query did
1005
+ // not match the deposited `eiffel tower country` — measured, that is the
1006
+ // trap's wrong answer.
1007
+ //
1008
+ // TAKEN FROM THE RECOGNITION THE RESPONSE ALREADY COMPUTED — the host's
1009
+ // `recogniseSpan`, the same surface the rest of this search uses — not from
1010
+ // a second offset scan. `canonicalQueryNodes` re-derived, per byte offset,
1011
+ // what `recognise` had already resolved once per query (its memo is keyed by
1012
+ // content), and that scan was the largest single cost the DIANOT join added:
1013
+ // measured against the pre-change tree, the same fixture and the same test
1014
+ // were 21 s slower with the scan than without it. A recognised site IS a
1015
+ // canonical node of the query that can lead somewhere, which is exactly the
1016
+ // set this filter wants, and it costs nothing to read.
1017
+ const queryNodes = new Set((this.host.recogniseSpan?.(queryBytes)?.sites ?? []).map((s) => s.payload));
1018
+ // TWO SOURCES, ONE ADMISSION. The recognition of a STORED WHOLE returns the
1019
+ // whole and stops — measured: for `The director of Eva is Gustaf Molander.`
1020
+ // it yields exactly ONE site, the fact's own node — so the entity a join
1021
+ // exists for is never proposed. The canonical fold is the second source,
1022
+ // and the scan runs only for a FORM (≥ W: a one-byte out is not something to
1023
+ // join through, and running it per letter measured 20-26 s in test/99).
1024
+ const W = this.maxGroup;
1025
+ const proposed = new Map();
1026
+ // The SOURCE of each proposal travels with it: a refusal that names only the
1027
+ // bytes leaves the next reader guessing which path proposed them — three
1028
+ // attempts at the chained join were spent fixing paths that never produced
1029
+ // the offending candidate.
1030
+ const source = new Map();
1031
+ for (const s of factRec.sites) {
1032
+ if (s.payload >= 0 && leads(s.payload)) {
1033
+ proposed.set(s.payload, this.store.bytesPrefix(s.payload, ALL));
1034
+ source.set(s.payload, "recognised site");
1035
+ }
1036
+ }
1037
+ if (this.host.canonResolve !== undefined && fact.bytes.length >= W) {
1038
+ const canon = this.host.canonResolve.bind(this.host);
1039
+ for (let start = 0; start < fact.bytes.length; start++) {
1040
+ for (let end = fact.bytes.length; end - start >= W; end--) {
1041
+ const id = canon(fact.bytes.subarray(start, end));
1042
+ if (id === null)
1043
+ continue;
1044
+ if (leads(id)) {
1045
+ proposed.set(id, this.store.bytesPrefix(id, ALL));
1046
+ source.set(id, "canonical fold");
1047
+ }
1048
+ break; // the longest form at this offset wins
1049
+ }
1050
+ }
1051
+ }
1052
+ const leading = [...proposed]
1053
+ .filter(([payload]) => payload !== fact.node && !queryNodes.has(payload))
1054
+ .map(([payload, bytes]) => ({ payload, bytes }));
907
1055
  // …then prefer the entity the query did NOT name, and the MAXIMAL one. The
908
1056
  // join exists to reach the subject the query never wrote, so:
909
1057
  // • a candidate the query already contains is the query's OWN subject, and
@@ -914,30 +1062,93 @@ export class GraphSearch {
914
1062
  // introduces ("Timur" must not win over "Timur Bekmambetov").
915
1063
  // Byte work over bytes already read, and the pruning REMOVES the
916
1064
  // resolve()/nextFirst() probes these candidates would have paid.
1065
+ if (leading.length === 0) {
1066
+ if (this.host.meter)
1067
+ this.host.meter.joinNoEntity++;
1068
+ if (reportable) {
1069
+ // Report WHAT the recognition returned, not just that nothing led: the
1070
+ // count and the first few site texts are the difference between "the
1071
+ // fact was not recognised" and "it was recognised but nothing led".
1072
+ const seen = factRec.sites.slice(0, 3).map((s) => this.store.bytesPrefix(s.payload, ALL));
1073
+ this.host.reportSearch?.("deriveThroughMiss", [fact.bytes, tail, ...seen], `no entity inside the fact leads anywhere — ${factRec.sites.length} site(s) recognised inside it`);
1074
+ }
1075
+ }
917
1076
  const candidates = leading
918
- .map((s) => ({
919
- payload: s.payload,
920
- bytes: this.store.bytesPrefix(s.payload, ALL),
921
- }))
922
- .filter((c) => indexOf(queryBytes, c.bytes, 0) < 0)
923
1077
  .filter((c, _i, all) => !all.some((o) => o.bytes.length > c.bytes.length && indexOf(o.bytes, c.bytes, 0) >= 0));
924
1078
  for (const c of candidates) {
925
- const key = this.host.resolve(concat2(c.bytes, tail));
926
- if (key === null)
927
- continue;
928
- const nx = this.store.nextFirst(key, 1);
929
- if (nx.length === 0)
1079
+ // THE SHORTEST TAIL PREFIX WHOSE KEY ALSO LEADS SOMEWHERE.
1080
+ //
1081
+ // The conclusion covers only that prefix, so the rest of the tail stays
1082
+ // for the step after — which is what CHAINING is (a whole-tail key
1083
+ // consumed the whole remainder and made every join terminal: measured,
1084
+ // `joinFired=0` on a three-relation query).
1085
+ //
1086
+ // A KEY THAT RESOLVES IS NOT ENOUGH. Measured with a dry run of these
1087
+ // very primitives: for the candidate `Sweden` the first tail prefix that
1088
+ // resolves is `" "` — the key `Sweden ` (a trailing space) — and it leads
1089
+ // NOWHERE (`nextFirst` = 0), so accepting it refused the join while the
1090
+ // key that names the fact (`sweden capital`) sat one prefix further. The
1091
+ // loop therefore asks BOTH questions before accepting, and keeps looking
1092
+ // otherwise.
1093
+ //
1094
+ // EXACT FIRST, THEN CANONICAL: the corpus holds both identities.
1095
+ let key = null;
1096
+ let used = 0;
1097
+ // The continuation the accepted key leads to, carried out of the loop:
1098
+ // the loop already HAD to read it to accept the key (a key that leads
1099
+ // nowhere is not the relation), so re-reading it after the loop was a
1100
+ // duplicate read and a branch that could never be taken.
1101
+ let next = null;
1102
+ let keyBytes = c.bytes;
1103
+ // THE PREFIX ENDS ARE THE TAIL'S OWN FOLD BOUNDARIES, not every byte
1104
+ // length. The key is `entity + prefix`, and the prefix that names a
1105
+ // stored relation ends where the fold cuts: measured over four join-firing
1106
+ // queries, 5 of 5 accepted keys ended on a boundary (or the tail's end)
1107
+ // while the byte-by-byte scan spent 153 probes where 14 boundaries would
1108
+ // do. Same criterion — resolves AND leads — same shortest-first order, so
1109
+ // the answer is the same one the enumeration found; only the candidates
1110
+ // come from the structure instead of from the byte count. A host with no
1111
+ // boundary rule falls back to the enumeration.
1112
+ const cuts = this.host.contentCuts?.(tail);
1113
+ const ends = cuts && cuts.length > 0
1114
+ ? [...cuts.filter((c) => c > 0 && c < tail.length), tail.length]
1115
+ : Array.from({ length: tail.length }, (_, i) => i + 1);
1116
+ for (const len of ends) {
1117
+ keyBytes = concat2(c.bytes, tail.subarray(0, len));
1118
+ const k = this.host.resolve(keyBytes) ??
1119
+ this.host.canonResolve?.(keyBytes) ??
1120
+ null;
1121
+ if (k === null)
1122
+ continue;
1123
+ const nx = this.store.nextFirst(k, 1);
1124
+ if (nx.length === 0)
1125
+ continue;
1126
+ key = k;
1127
+ used = len;
1128
+ next = nx[0];
1129
+ break;
1130
+ }
1131
+ if (key === null) {
1132
+ if (this.host.meter)
1133
+ this.host.meter.joinNoKey++;
1134
+ if (reportable) {
1135
+ this.host.reportSearch?.("deriveThroughMiss", [c.bytes, tail, keyBytes], `no learnt key names this entity and tail together ` +
1136
+ `(candidate #${c.payload}, from the ${source.get(c.payload) ?? "unknown"} source)`);
1137
+ }
930
1138
  continue;
1139
+ }
1140
+ if (this.host.meter)
1141
+ this.host.meter.joinFired++;
931
1142
  yield {
932
1143
  premises: [fact],
933
1144
  conclusion: {
934
1145
  kind: "out",
935
1146
  i: fact.i,
936
- j: queryLen,
937
- bytes: this.store.bytesPrefix(nx[0], ALL),
1147
+ j: fact.j + used,
1148
+ bytes: this.store.bytesPrefix(next, ALL),
938
1149
  cover: true,
939
1150
  rec: true,
940
- node: nx[0],
1151
+ node: next,
941
1152
  throughFact: true,
942
1153
  },
943
1154
  cost: STEP,
@@ -1,5 +1,5 @@
1
1
  export { Mind } from "./mind.js";
2
- export type { Input, Response } from "./mind.js";
2
+ export type { CorpusTextPair, CorpusTextResult, Input, Response, } from "./mind.js";
3
3
  export type { ComputedSpan, ExtensionHost } from "./mind.js";
4
4
  export type { MechanismResult, PipelineMechanism, Precomputed, } from "./pipeline-mechanism.js";
5
5
  export type { InspectRationale, RationaleItem, RationaleStep, } from "./rationale.js";
@@ -7,3 +7,5 @@ export type { AnchorRejectionReason, ClimbConsensusData, ConsensusAnchorTrace, C
7
7
  export type { AncestorReach, AttentionRead, SaturationReason, SaturationStop, } from "./types.js";
8
8
  export type { DepositReport } from "./learning.js";
9
9
  export type { DecideGroundingData, NarrowDecisionData, Provenance, } from "./pipeline.js";
10
+ export { sampleCorpus, searchCorpus } from "./corpus.js";
11
+ export type { CorpusMiss, CorpusPair, CorpusResult } from "./corpus.js";
@@ -3,3 +3,4 @@
3
3
  // Re-exports the Mind class and all public types that were previously
4
4
  // exported from mind/mind.ts directly.
5
5
  export { Mind } from "./mind.js";
6
+ export { sampleCorpus, searchCorpus } from "./corpus.js";
@@ -78,9 +78,14 @@ export interface AlignGap {
78
78
  }
79
79
  /** Extend a seed match (query offset qo ↔ candidate offset co) to its maximal
80
80
  * common run, then walk outward in both directions collecting further common
81
- * runs of at least W bytes across bounded mismatch gaps (each side ≤
82
- * chainReach). Returns the matched query spans and the mismatch pairs
83
- * between consecutive runs.
81
+ * runs of at least W bytes across mismatch gaps. Each gap's LENGTH is the
82
+ * pair's own extent (a gap cannot be longer than the bytes it spans) and the
83
+ * sweep's WORK is proportional to the bytes a run spans (the context's windows
84
+ * are indexed once, then the query's are walked) — the arity bound
85
+ * (`chainReach`) used to cap BOTH, and truncated every learned frame whose
86
+ * slot was longer. Each sweep owns its own budget, so an exhausted right
87
+ * sweep never starves the left one. Returns the matched query spans and the
88
+ * mismatch pairs between consecutive runs.
84
89
  *
85
90
  * This is the SEEDED aligner, distinct from {@link alignRuns}: that one finds
86
91
  * every run two structures share anywhere (a weave), this one reads two
@@ -31,8 +31,8 @@
31
31
  // they gate.
32
32
  import { addInto, cosine, dot, normalize, zeros } from "../vec.js";
33
33
  import { conceptThreshold, dominates, identityBar, significanceBar, } from "../geometry.js";
34
- import { bytesEqual, indexOf } from "../bytes.js";
35
- import { chainReach, leafIdRun } from "./canonical.js";
34
+ import { bytesEqual, indexOf, latin1 } from "../bytes.js";
35
+ import { leafIdRun } from "./canonical.js";
36
36
  import { foldTree, gistOf, perceive, read, resolve } from "./primitives.js";
37
37
  import { argmaxCosine, chooseAmong, chooseNext, corpusN, edgeAncestors, guidedFirst, hubBound, hubCap, sharedReachMemo, } from "./traverse.js";
38
38
  import { recognise, segment } from "./recognition.js";
@@ -237,9 +237,14 @@ export function alignGraded(ctx, query, contextBytes, querySites) {
237
237
  }
238
238
  /** Extend a seed match (query offset qo ↔ candidate offset co) to its maximal
239
239
  * common run, then walk outward in both directions collecting further common
240
- * runs of at least W bytes across bounded mismatch gaps (each side ≤
241
- * chainReach). Returns the matched query spans and the mismatch pairs
242
- * between consecutive runs.
240
+ * runs of at least W bytes across mismatch gaps. Each gap's LENGTH is the
241
+ * pair's own extent (a gap cannot be longer than the bytes it spans) and the
242
+ * sweep's WORK is proportional to the bytes a run spans (the context's windows
243
+ * are indexed once, then the query's are walked) — the arity bound
244
+ * (`chainReach`) used to cap BOTH, and truncated every learned frame whose
245
+ * slot was longer. Each sweep owns its own budget, so an exhausted right
246
+ * sweep never starves the left one. Returns the matched query spans and the
247
+ * mismatch pairs between consecutive runs.
243
248
  *
244
249
  * This is the SEEDED aligner, distinct from {@link alignRuns}: that one finds
245
250
  * every run two structures share anywhere (a weave), this one reads two
@@ -253,7 +258,21 @@ export function alignGraded(ctx, query, contextBytes, querySites) {
253
258
  * window test); {@link frameSlots} takes the other reading. */
254
259
  export function alignAround(ctx, q, c, qo, co) {
255
260
  const W = ctx.space.maxGroup;
256
- const reachCap = chainReach(W);
261
+ // THE GAP LENGTH IS THE PAIR'S OWN EXTENT; THE WORK IS BUDGETED.
262
+ //
263
+ // The sweep walks (queryGap, contextGap) pairs by ASCENDING total, so reaching
264
+ // a gap of size G costs about G²/2 pairs. Bounding the LENGTH by the write
265
+ // side's arity (`chainReach(W)` = 16) therefore truncated every learned frame
266
+ // whose slot is longer — measured: `bindReference` reported the cap at 18, 24,
267
+ // 30 and 36 bytes and `recall` answered with ANOTHER instance's filler — while
268
+ // removing the bound outright took the corpus-cost guard (test/89) from
269
+ // milliseconds to 68 seconds.
270
+ //
271
+ // Bounding the PAIRS keeps a call's cost constant however long the pair is,
272
+ // and the ascending order means an exhausted budget drops the FAR
273
+ // continuations and never the near ones — the same degradation recognition.ts
274
+ // documents for its canon budget. Length and work are different questions;
275
+ // this is the one place they were conflated.
257
276
  // Maximal run around the seed.
258
277
  let qs = qo, ss = co;
259
278
  while (qs > 0 && ss > 0 && q[qs - 1] === c[ss - 1]) {
@@ -267,8 +286,66 @@ export function alignAround(ctx, q, c, qo, co) {
267
286
  }
268
287
  const matched = [[qs, qe]];
269
288
  const gaps = [];
270
- // The next common run of ≥ W bytes past (qi, si), with each side's gap
271
- // bounded by chainReach; smallest total gap wins (nearest continuation).
289
+ // THE SWEEP IS STRUCTURAL, NOT ENUMERATIVE.
290
+ //
291
+ // The criterion is unchanged: the next common run, MINIMUM TOTAL GAP, ties to
292
+ // the smaller query gap. What changed is how it is found. Enumerating
293
+ // (queryGap, contextGap) pairs by ascending total reaches a run at total t in
294
+ // about t²/2 pairs — and that quadratic shape, not the reach, was the cost
295
+ // problem: capping the pairs dropped reach (a legitimate 24-byte slot stopped
296
+ // being found), while leaving them uncapped cost 68 seconds on the corpus
297
+ // guard. Neither is the answer, because the answer is the algorithm.
298
+ //
299
+ // The context's windows are indexed ONCE, for lengths 1..W — W being the
300
+ // geometry's own unit of composition, so nothing is chosen here. Each step
301
+ // then walks the query's windows outward from the anchor: for a given query
302
+ // gap the nearest context gap that continues a run is one O(1) lookup, and the
303
+ // walk stops the moment the query gap alone exceeds the best total already
304
+ // found. So the work is proportional to the bytes the run SPANS. No budget,
305
+ // no cap, no number: a long slot is reached, and its price is already the
306
+ // ladder's (its bytes are unaccounted, so the search pays PASS per byte).
307
+ const index = [];
308
+ for (let len = 1; len <= W; len++) {
309
+ const m = new Map();
310
+ for (let o = 0; o + len <= c.length; o++) {
311
+ const key = latin1(c.subarray(o, o + len));
312
+ const at = m.get(key);
313
+ if (at === undefined)
314
+ m.set(key, [o]);
315
+ else
316
+ at.push(o);
317
+ }
318
+ index.push(m);
319
+ }
320
+ /** Smallest listed offset at or after `from`, or -1. */
321
+ const fromAt = (list, from) => {
322
+ let lo = 0, hi = list.length - 1, best = -1;
323
+ while (lo <= hi) {
324
+ const mid = (lo + hi) >> 1;
325
+ if (list[mid] >= from) {
326
+ best = list[mid];
327
+ hi = mid - 1;
328
+ }
329
+ else
330
+ lo = mid + 1;
331
+ }
332
+ return best;
333
+ };
334
+ /** Largest listed offset at or before `to`, or -1. */
335
+ const toAt = (list, to) => {
336
+ let lo = 0, hi = list.length - 1, best = -1;
337
+ while (lo <= hi) {
338
+ const mid = (lo + hi) >> 1;
339
+ if (list[mid] <= to) {
340
+ best = list[mid];
341
+ lo = mid + 1;
342
+ }
343
+ else
344
+ hi = mid - 1;
345
+ }
346
+ return best;
347
+ };
348
+ /** Length of the common run STARTING at (qi, si). */
272
349
  const runLenAt = (qi, si) => {
273
350
  let n = 0;
274
351
  while (qi + n < q.length && si + n < c.length && q[qi + n] === c[si + n]) {
@@ -276,69 +353,76 @@ export function alignAround(ctx, q, c, qo, co) {
276
353
  }
277
354
  return n;
278
355
  };
279
- // RIGHT sweep.
280
- let qi = qe, si = se;
281
- for (;;) {
282
- let found = false;
283
- for (let total = 1; total <= 2 * reachCap && !found; total++) {
284
- for (let gq = 0; gq <= Math.min(total, reachCap); gq++) {
285
- const gs = total - gq;
286
- if (gs > reachCap)
356
+ /** Length of the common run ENDING at (qi, si). */
357
+ const runLenBefore = (qi, si) => {
358
+ let n = 0;
359
+ while (n < qi && n < si && q[qi - 1 - n] === c[si - 1 - n])
360
+ n++;
361
+ return n;
362
+ };
363
+ /** The next run outward from an anchor, or null when the bytes run out. */
364
+ const nextRun = (qi, si, forward) => {
365
+ const qLim = forward ? q.length - qi : qi;
366
+ let best = null;
367
+ for (let gq = 0; gq < qLim; gq++) {
368
+ // No later query gap can beat a total already found.
369
+ if (best !== null && gq > best.gq + best.gs)
370
+ break;
371
+ const left = qLim - gq;
372
+ // A run of >= W bytes, or — when the query itself ends inside one window —
373
+ // the run that REACHES that end. Exactly the acceptance the sweep had.
374
+ const lens = left >= W ? [W] : [left];
375
+ for (const len of lens) {
376
+ const key = latin1(q.subarray(forward ? qi + gq : qi - gq - len, forward ? qi + gq + len : qi - gq));
377
+ const list = index[len - 1].get(key);
378
+ if (list === undefined)
287
379
  continue;
288
- if (qi + gq >= q.length || si + gs >= c.length)
380
+ const o = forward ? fromAt(list, si) : toAt(list, si - len);
381
+ if (o < 0)
289
382
  continue;
290
- const n = runLenAt(qi + gq, si + gs);
291
- if (n >= W || qi + gq + n === q.length) {
292
- if (n === 0)
293
- continue;
294
- if (gq > 0 || gs > 0) {
295
- gaps.push({ qs: qi, qe: qi + gq, cs: si, ce: si + gs });
383
+ const n = forward
384
+ ? runLenAt(qi + gq, o)
385
+ : runLenBefore(qi - gq, o + len);
386
+ if (n < 1)
387
+ continue;
388
+ if (forward ? n >= W || qi + gq + n === q.length : n >= W || n === qi - gq) {
389
+ const gs = forward ? o - si : si - len - o;
390
+ if (best === null || gq + gs < best.gq + best.gs) {
391
+ best = { gq, gs, n };
296
392
  }
297
- matched.push([qi + gq, qi + gq + n]);
298
- qi = qi + gq + n;
299
- si = si + gs + n;
300
- found = true;
301
393
  break;
302
394
  }
303
395
  }
304
396
  }
305
- if (!found)
397
+ return best;
398
+ };
399
+ // RIGHT sweep.
400
+ let qi = qe, si = se;
401
+ for (;;) {
402
+ const step = nextRun(qi, si, true);
403
+ if (step === null)
306
404
  break;
405
+ if (step.gq > 0 || step.gs > 0) {
406
+ gaps.push({ qs: qi, qe: qi + step.gq, cs: si, ce: si + step.gs });
407
+ }
408
+ matched.push([qi + step.gq, qi + step.gq + step.n]);
409
+ qi = qi + step.gq + step.n;
410
+ si = si + step.gs + step.n;
307
411
  }
308
- // LEFT sweep (mirror).
412
+ // LEFT sweep (mirror): an independent walk, so an exhausted right side can
413
+ // never starve it (pinned by test/114).
309
414
  qi = qs;
310
415
  si = ss;
311
416
  for (;;) {
312
- let found = false;
313
- for (let total = 1; total <= 2 * reachCap && !found; total++) {
314
- for (let gq = 0; gq <= Math.min(total, reachCap); gq++) {
315
- const gs = total - gq;
316
- if (gs > reachCap)
317
- continue;
318
- if (qi - gq <= 0 || si - gs <= 0)
319
- continue;
320
- // Run ENDING at (qi - gq, si - gs).
321
- let n = 0;
322
- while (n < qi - gq && n < si - gs &&
323
- q[qi - gq - 1 - n] === c[si - gs - 1 - n]) {
324
- n++;
325
- }
326
- if (n >= W || n === qi - gq) {
327
- if (n === 0)
328
- continue;
329
- if (gq > 0 || gs > 0) {
330
- gaps.push({ qs: qi - gq, qe: qi, cs: si - gs, ce: si });
331
- }
332
- matched.push([qi - gq - n, qi - gq]);
333
- qi = qi - gq - n;
334
- si = si - gs - n;
335
- found = true;
336
- break;
337
- }
338
- }
339
- }
340
- if (!found)
417
+ const step = nextRun(qi, si, false);
418
+ if (step === null)
341
419
  break;
420
+ if (step.gq > 0 || step.gs > 0) {
421
+ gaps.push({ qs: qi - step.gq, qe: qi, cs: si - step.gs, ce: si });
422
+ }
423
+ matched.push([qi - step.gq - step.n, qi - step.gq]);
424
+ qi = qi - step.gq - step.n;
425
+ si = si - step.gs - step.n;
342
426
  }
343
427
  return { matched, gaps };
344
428
  }