@hviana/sema 0.8.1 → 0.8.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (114) hide show
  1. package/AGENTS.md +29 -29
  2. package/TRADEMARKS.md +0 -1
  3. package/dist/src/config.d.ts +28 -0
  4. package/dist/src/config.js +20 -0
  5. package/dist/src/geometry.d.ts +21 -0
  6. package/dist/src/geometry.js +21 -0
  7. package/dist/src/meter.d.ts +76 -0
  8. package/dist/src/meter.js +95 -0
  9. package/dist/src/mind/attention.d.ts +4 -0
  10. package/dist/src/mind/attention.js +165 -16
  11. package/dist/src/mind/canonical.d.ts +16 -0
  12. package/dist/src/mind/canonical.js +41 -0
  13. package/dist/src/mind/corpus.d.ts +40 -0
  14. package/dist/src/mind/corpus.js +149 -0
  15. package/dist/src/mind/graph-search.d.ts +7 -0
  16. package/dist/src/mind/graph-search.js +254 -24
  17. package/dist/src/mind/index.d.ts +3 -1
  18. package/dist/src/mind/index.js +1 -0
  19. package/dist/src/mind/match.d.ts +9 -4
  20. package/dist/src/mind/match.js +147 -61
  21. package/dist/src/mind/mechanisms/cast.js +19 -3
  22. package/dist/src/mind/mechanisms/confluence.js +24 -0
  23. package/dist/src/mind/mechanisms/cover.js +6 -0
  24. package/dist/src/mind/mechanisms/recall.js +32 -4
  25. package/dist/src/mind/mind.d.ts +57 -0
  26. package/dist/src/mind/mind.js +72 -1
  27. package/dist/src/mind/pipeline-mechanism.d.ts +7 -0
  28. package/dist/src/mind/pipeline.js +66 -20
  29. package/dist/src/mind/primitives.js +9 -1
  30. package/dist/src/mind/rationale.d.ts +28 -1
  31. package/dist/src/mind/rationale.js +22 -1
  32. package/dist/src/mind/reasoning.d.ts +25 -3
  33. package/dist/src/mind/reasoning.js +125 -20
  34. package/dist/src/mind/recognition.js +4 -8
  35. package/dist/src/mind/resonance.js +20 -1
  36. package/dist/src/mind/trace.js +1 -0
  37. package/dist/src/mind/traverse.js +15 -3
  38. package/dist/src/mind/types.d.ts +49 -4
  39. package/docs/INVARIANTS.md +2 -2
  40. package/docs/architecture/bounded-reads.md +1 -1
  41. package/docs/architecture/commonality.md +2 -2
  42. package/docs/architecture/cost-model.md +2 -2
  43. package/docs/architecture/determinism.md +7 -7
  44. package/docs/architecture/match-project.md +2 -3
  45. package/docs/architecture/mechanism-market.md +10 -10
  46. package/docs/architecture/meter.md +5 -5
  47. package/docs/architecture/store.md +3 -3
  48. package/docs/failures/tempting-but-wrong.md +34 -6
  49. package/docs/harness/gates.md +2 -2
  50. package/docs/mechanisms/cast.md +2 -2
  51. package/docs/mechanisms/cover.md +2 -3
  52. package/docs/mechanisms/extraction.md +7 -7
  53. package/docs/mechanisms/recall.md +8 -9
  54. package/jsr.json +1 -1
  55. package/package.json +1 -1
  56. package/src/alu/README.md +11 -12
  57. package/src/config.ts +48 -0
  58. package/src/geometry.ts +21 -0
  59. package/src/meter.ts +98 -0
  60. package/src/mind/attention.ts +167 -16
  61. package/src/mind/canonical.ts +43 -0
  62. package/src/mind/corpus.ts +202 -0
  63. package/src/mind/graph-search.ts +277 -23
  64. package/src/mind/index.ts +8 -1
  65. package/src/mind/match.ts +148 -57
  66. package/src/mind/mechanisms/cast.ts +20 -2
  67. package/src/mind/mechanisms/confluence.ts +24 -0
  68. package/src/mind/mechanisms/cover.ts +5 -0
  69. package/src/mind/mechanisms/recall.ts +32 -4
  70. package/src/mind/mind.ts +125 -0
  71. package/src/mind/pipeline-mechanism.ts +7 -0
  72. package/src/mind/pipeline.ts +79 -22
  73. package/src/mind/primitives.ts +9 -1
  74. package/src/mind/rationale.ts +35 -1
  75. package/src/mind/reasoning.ts +145 -13
  76. package/src/mind/recognition.ts +4 -8
  77. package/src/mind/resonance.ts +19 -1
  78. package/src/mind/trace.ts +1 -0
  79. package/src/mind/traverse.ts +16 -6
  80. package/src/mind/types.ts +53 -4
  81. package/test/100-complete-grounding-trace.test.mjs +109 -0
  82. package/test/101-alignment-gap-bound.test.mjs +106 -0
  83. package/test/102-production-composes-at-scale.test.mjs +110 -0
  84. package/test/103-alignment-gap-budget.test.mjs +89 -0
  85. package/test/104-composition-is-reported.test.mjs +90 -0
  86. package/test/105-derive-through-reports-its-refusal.test.mjs +137 -0
  87. package/test/106-the-join-fires.test.mjs +94 -0
  88. package/test/107-the-join-is-counted.test.mjs +81 -0
  89. package/test/108-the-join-chains.test.mjs +78 -0
  90. package/test/109-the-pivot-is-counted.test.mjs +60 -0
  91. package/test/110-the-reasoner-stops-when-the-question-is-answered.test.mjs +91 -0
  92. package/test/111-the-cover-assembly-is-counted.test.mjs +74 -0
  93. package/test/112-the-exploration-does-not-grow-with-the-hub.test.mjs +89 -0
  94. package/test/113-the-rationale-payload-is-bounded.test.mjs +84 -0
  95. package/test/114-alignment-budget-is-per-sweep.test.mjs +93 -0
  96. package/test/116-the-extension-is-gated-by-the-pipelines-own-remainder.test.mjs +100 -0
  97. package/test/117-corpus-search.test.mjs +171 -0
  98. package/test/118-the-join-reaches-a-key-off-the-cut.test.mjs +74 -0
  99. package/test/119-the-work-does-not-grow-with-the-corpus.test.mjs +122 -0
  100. package/test/120-composition-is-consequence.test.mjs +132 -0
  101. package/test/121-the-extension-does-not-grow-with-the-corpus.test.mjs +128 -0
  102. package/test/122-the-climb-search-does-not-grow-with-the-corpus.test.mjs +117 -0
  103. package/test/123-the-paired-formulas-agree.test.mjs +90 -0
  104. package/test/125-the-post-grounding-branch-publishes-its-operand.test.mjs +51 -0
  105. package/test/126-the-pipeline-does-not-name-mechanisms.test.mjs +42 -0
  106. package/test/128-the-leads-somewhere-pair-agrees.test.mjs +83 -0
  107. package/test/129-the-trace-payload-shape.test.mjs +164 -0
  108. package/test/14-scaling.test.mjs +10 -7
  109. package/test/32-confluence.test.mjs +68 -0
  110. package/test/38-reason-restate-guard.test.mjs +8 -2
  111. package/test/43-cast-analog-seat.test.mjs +10 -0
  112. package/test/55-cost-meter.test.mjs +859 -0
  113. package/test/76-reference-binding.test.mjs +6 -1
  114. package/test/89-completion-recursion.test.mjs +30 -5
@@ -775,15 +775,18 @@ export async function voteRegions(ctx, query, regions, k, mode, N, reachMemo, td
775
775
  }
776
776
  contrastiveMargin = margin;
777
777
  // Scaled by what this region does NOT address — see `cov` above.
778
- const noiseFloor = estimatorNoise(ctx.store.D) * (1 - cov);
779
- if (margin <= noiseFloor) {
778
+ // The bar THIS gate applies: the estimator's noise scaled by what the
779
+ // region does NOT address (`cov`). ONE definition, used by the rejection
780
+ // path below and by the voted payload — the trace reports the applied bar.
781
+ const appliedFloor = estimatorNoise(ctx.store.D) * (1 - cov);
782
+ if (margin <= appliedFloor) {
780
783
  recordRegion("contrastive-margin-rejection", {
781
784
  selected,
782
785
  reachNode: voterId,
783
786
  idf,
784
787
  dfWeight: wf,
785
788
  contrastiveMargin: margin,
786
- contrastiveNoiseFloor: noiseFloor,
789
+ contrastiveNoiseFloor: appliedFloor,
787
790
  ...(contrastiveRival ? { contrastiveRival } : {}),
788
791
  });
789
792
  continue;
@@ -834,7 +837,12 @@ export async function voteRegions(ctx, query, regions, k, mode, N, reachMemo, td
834
837
  ...(contrastiveMargin !== undefined
835
838
  ? {
836
839
  contrastiveMargin,
837
- contrastiveNoiseFloor: estimatorNoise(ctx.store.D),
840
+ // THE BAR THE GATE ACTUALLY APPLIED — the same expression
841
+ // the rejection path's `appliedFloor` defines, inline here because
842
+ // this payload is built in a scope that does not carry that local.
843
+ // Publishing the raw estimatorNoise(D) instead made a region that
844
+ // PASSED look closer to its limit than it was.
845
+ contrastiveNoiseFloor: estimatorNoise(ctx.store.D) * (1 - cov),
838
846
  ...(contrastiveRival ? { contrastiveRival } : {}),
839
847
  }
840
848
  : {}),
@@ -939,7 +947,17 @@ export function poolVotes(ctx, regionVotes, sat, N, td) {
939
947
  },
940
948
  pool,
941
949
  };
942
- lightestDerivation(system);
950
+ // THE SEARCH WAS THE ONE LAYER WITH NO TIME. The climb's phases are timed
951
+ // (voteRegions, structuralResonance, crossRegion) but the pooled derivation
952
+ // was not, so any cost or gain inside it stayed invisible.
953
+ // `timeSync`, not `time`: the search is SYNCHRONOUS, and wrapping it in a
954
+ // promise only to time it would make the profiled path wait where the
955
+ // unprofiled one does not (meter.ts's own contract).
956
+ if (ctx.meter) {
957
+ ctx.meter.timeSync("climb.derivation", () => lightestDerivation(system));
958
+ }
959
+ else
960
+ lightestDerivation(system);
943
961
  const votes = new Map();
944
962
  const votesIdf = new Map();
945
963
  const support = new Map();
@@ -973,12 +991,24 @@ export function poolVotes(ctx, regionVotes, sat, N, td) {
973
991
  // The LARGEST single region's contribution to this anchor's pooled vote.
974
992
  // The pool is a SUM (deliberately — see the pooling note above), so it says
975
993
  // how much evidence there is in total, never whether any ONE place in the
976
- // query carries evidence on its own. Consumers that hold an anchor to
977
- // consensusFloor(N) = ln(N) + 1/2 need the latter: that bar prices ONE
978
- // region's maximally-discriminative evidence (ln N is the IDF of content
979
- // reaching a single context), so comparing a six-region sum against it is a
980
- // dimensional error. Recorded here, beside the count, because this is the
981
- // only place the per-region contributions are still separable.
994
+ // query carries evidence on its own. Recorded here, beside the count,
995
+ // because this is the only place the per-region contributions are still
996
+ // separable.
997
+ //
998
+ // THE BAR IS THE POOLED FLOOR, AND IT WAS ONCE CLAIMED OTHERWISE HERE.
999
+ // This comment used to say that holding an anchor to consensusFloor(N)
1000
+ // "prices ONE region's evidence", so comparing a six-region sum against it
1001
+ // was "a dimensional error". THAT WAS FALSE. `thresholds.md` §2 derives
1002
+ // `consensusFloor` as the POOLED-vote significance floor ("each region
1003
+ // contributes at most ln(N/c) <= ln(N); ln(N) + 1/2 demands ..."), and the
1004
+ // climb weights by IDF, so the sum and the floor are in ONE dimension —
1005
+ // which is exactly why `recall.ts` gates `forest[0].idfVote` against it and
1006
+ // why `commitVotes` does too. The other two weighting modes DO leave that
1007
+ // dimension (by at most ln 2, two-sided: `direct` deflates a region and
1008
+ // `combined` inflates it), and the gates therefore read the IDF sum, which
1009
+ // is mode-independent; `test/55` tests 19 and 20 pin both halves — the sum
1010
+ // as the reading the bar is derived for, and the absence of any gate
1011
+ // inversion across the three modes.
982
1012
  const regionPeak = new Map();
983
1013
  const steps = [];
984
1014
  let order = 0;
@@ -1136,12 +1166,23 @@ export function commitVotes(ctx, pooled, sat, regions, regionVoter, N, td, cfg)
1136
1166
  anchor,
1137
1167
  vote,
1138
1168
  peak: regionPeak.get(anchor) ?? 0,
1169
+ idfVote: votesIdf.get(anchor) ?? 0,
1139
1170
  start: s.start,
1140
1171
  end: s.end,
1141
1172
  breadth: (regionSupport.get(anchor) ?? 0) / totalRegions,
1142
1173
  clusters: countClusters(regionSpans.get(anchor) ?? [], ctx.space.maxGroup),
1143
1174
  };
1144
1175
  })
1176
+ // THE ORDER IS NOT A PREFERENCE: with equal evidence it decides ADMISSION,
1177
+ // through the stable sort and the first-come overlap absorption below.
1178
+ // Measured on test/34's corpus, query "red": the two candidates (`red
1179
+ // circle` and `red square`) carry IDENTICAL `vote` and IDENTICAL `idfVote`
1180
+ // (1.3863 each, three seeds), so this comparator leaves them tied and the
1181
+ // stable sort keeps the ENUMERATION order — which is corpus-determined and
1182
+ // admits `red square`, 60/60 seeds. Adding an id tie-break (`|| a.anchor -
1183
+ // b.anchor`) picks `red circle` instead and makes a single region reach the
1184
+ // JOINT context, which is the premise `test/34` exists to protect. The
1185
+ // gates read IDF; this line only decides who gets looked at first.
1145
1186
  .sort((a, b) => b.vote - a.vote);
1146
1187
  const overlaps = (a, b) => a.start < b.end && b.start < a.end;
1147
1188
  // Read the root cut from the anchors the QUERY pointed at. A vote standing
@@ -1177,6 +1218,13 @@ export function commitVotes(ctx, pooled, sat, regions, regionVoter, N, td, cfg)
1177
1218
  rank,
1178
1219
  pooledVote: point.vote,
1179
1220
  idfVote: votesIdf.get(point.anchor) ?? 0,
1221
+ // The LARGEST single-region contribution behind this anchor — the bar
1222
+ // recall's own gate reads (mechanisms/recall.ts: forest[0].peak > LN2),
1223
+ // and until now the only decision-making quantity the climb computed and
1224
+ // did not publish. `regionPeak` reached `ranked` (see its build below)
1225
+ // and stopped there. Published, not recomputed: the value is the one the
1226
+ // climb already carries.
1227
+ peak: point.peak,
1180
1228
  candidateBreadth: regions.length,
1181
1229
  contributingVotes: regionAxioms.get(point.anchor) ?? 0,
1182
1230
  contributingEvidence: regionSupport.get(point.anchor) ?? 0,
@@ -1206,6 +1254,18 @@ export function commitVotes(ctx, pooled, sat, regions, regionVoter, N, td, cfg)
1206
1254
  let passesConsensusFloor;
1207
1255
  let pastLeadingSaturation;
1208
1256
  let tiedWithDominant;
1257
+ // ── ONE OF THREE ADMISSIONS, AND THEY ARE NOT THE SAME READING ────────
1258
+ // This block admits by VOTES: per-region evidence pooled, gated on the
1259
+ // natural break and on consensusFloor, with the dominant allowed to bypass
1260
+ // both. `structuralResonance` admits by a MARGIN over the estimator's own
1261
+ // noise, and `crossRegionVotes` admits by STRUCTURE (which regions may pair
1262
+ // at all, with at least one side individually discriminative). Read
1263
+ // together they look like one policy written three times; they are three
1264
+ // different measurements of the same question ("is this evidence?"), and
1265
+ // unifying them would average three readings into one — the mistake
1266
+ // `extraction.ts` records as "do not unify the two into one machine".
1267
+ // What they DO share, and must keep sharing, is the discipline of deriving
1268
+ // every bar from D/W/N rather than choosing it (thresholds.md).
1209
1269
  const rejectionReasons = [];
1210
1270
  if (absorbed) {
1211
1271
  status = "overlap";
@@ -1216,9 +1276,75 @@ export function commitVotes(ctx, pooled, sat, regions, regionVoter, N, td, cfg)
1216
1276
  pastLeadingSaturation = pastLeading;
1217
1277
  const vote = votesIdf.get(point.anchor) ?? 0;
1218
1278
  if (roots.length === 0) {
1219
- // The first non-overlapping root is DOMINANT and bypasses the two
1220
- // vote thresholds (it always grounds) — only the leading-saturation
1221
- // gate still applies to it.
1279
+ // THE DOMINANCE PRIVILEGE, AND THE TENSION IT CARRIES (measured).
1280
+ //
1281
+ // The first non-overlapping candidate is DOMINANT: it bypasses both
1282
+ // vote gates below and grounds on its own; only the leading-saturation
1283
+ // gate still applies to it. The privilege is load-bearing — analogies,
1284
+ // substitutions and composed contexts are precisely candidates the
1285
+ // query does NOT contain, and the engine loses them without it.
1286
+ //
1287
+ // WHICH candidate receives it, though, is decided by this loop's ORDER.
1288
+ // That order comes from `ranked`, and when two candidates carry equal
1289
+ // evidence the stable sort preserves the ENUMERATION order, so the
1290
+ // privilege is allocated by an ordering rather than by a rule.
1291
+ //
1292
+ // Measured on test/34's corpus, query "red":
1293
+ //
1294
+ // 0:#77 vote=1.3863 idf=1.3863 [0,3) | 1:#49 vote=1.3863 idf=1.3863 [0,3)
1295
+ //
1296
+ // Both candidates (`red square` #77, `red circle` #49) have IDENTICAL
1297
+ // `vote` AND IDENTICAL `idfVote` over the SAME support span, so the
1298
+ // comparator leaves them tied, the second is absorbed as "overlap", and
1299
+ // the first grounds. With this build's enumeration order that first is
1300
+ // `red square`, 60/60 seeds, and `red circle` — the JOINT context — is
1301
+ // never reached by "red" alone. That is the premise test/34 exists to
1302
+ // protect: no single region reaches the joint context, which is what
1303
+ // makes the binding query unreachable without direct region
1304
+ // interaction.
1305
+ //
1306
+ // THE TENSION: the premise therefore holds BY ENUMERATION ORDER, not by
1307
+ // a rule, so any change to this ordering can move the privilege onto the
1308
+ // joint context and let one region reach it. Measured: adding
1309
+ // `|| a.anchor - b.anchor` — the lowest-id tie-break that AGENTS.md §2
1310
+ // sanctions as an equivalent corpus-determined tie-break — does exactly
1311
+ // that: "red" then attends to `red circle`, test/34 fails 6/1, and the
1312
+ // canonical suite reports 1 failure.
1313
+ //
1314
+ // TWO ATTEMPTS TO MAKE IT A RULE, BOTH REFUTED BY MEASUREMENT:
1315
+ //
1316
+ // 1. EVIDENCE SEPARATION. Grant the privilege only when the first
1317
+ // candidate's evidence is separated from the next distinct
1318
+ // candidate's by more than the co-dominant band (sqrt(k) *
1319
+ // estimatorNoise(D)). Refuted: that band exists to ADMIT the
1320
+ // anchors the estimator cannot separate from the dominant — its own
1321
+ // documented purpose — so withholding the privilege on ties removes
1322
+ // the very case it was written for. Suite: 4 failures (the two
1323
+ // co-dominant band laws, breadth/scale invariance, test/29 D2).
1324
+ //
1325
+ // 2. QUERY-OWNED CONTENT. Grant the privilege only to a candidate
1326
+ // that IS a recognised region's identity (regions.some(r => r.id ===
1327
+ // point.anchor)). Measured: for "circle" that identity IS the
1328
+ // ranked candidate, so the privilege stays and `circle` grounds; for
1329
+ // "red" the identity is the `red` node itself while the candidates
1330
+ // are the conjunctions, so neither is privileged; for "red then
1331
+ // circle" the composed context carries idf 3.958 and clears both
1332
+ // gates on its own evidence. All four control queries came out
1333
+ // right — and the suite: 10 failures, six of them in the
1334
+ // analogy/counterfactual/CAST suites ("an analogy still transfers
1335
+ // from a structure the query never names"; "a substitute the query
1336
+ // NAMES may still be voiced"). Refuted: the privilege exists to
1337
+ // admit what the query does NOT contain, so identity is the wrong
1338
+ // axis.
1339
+ //
1340
+ // WHAT A FUTURE ATTEMPT MUST RESPECT: whatever allocates this privilege
1341
+ // has to (a) keep it available to candidates the query does not contain
1342
+ // — analogies, substitutions, compositions — and (b) not depend on the
1343
+ // estimator's ordering among anchors of equal evidence, because that
1344
+ // ordering is not a fact about the corpus. No lever satisfying both has
1345
+ // been found. Until one is, this premise rests on the enumeration order
1346
+ // recorded above, and test/34 is the only test that notices if it
1347
+ // moves.
1222
1348
  dominant = true;
1223
1349
  if (pastLeading) {
1224
1350
  status = "root";
@@ -1229,8 +1355,15 @@ export function commitVotes(ctx, pooled, sat, regions, regionVoter, N, td, cfg)
1229
1355
  }
1230
1356
  }
1231
1357
  else {
1232
- passesNaturalBreak = vote >= rootCut;
1233
- passesConsensusFloor = vote >= floor;
1358
+ // THE FLOOR AND THE BREAK READ THE IDF WEIGHTING. `floor` is derived
1359
+ // for pooled IDF-weighted votes, and `rootCut` comes from the IDF
1360
+ // distribution (`idfDesc`), so gating the mode-dependent `vote` against
1361
+ // either let a weighting mode change an admission (measured: anchor 87,
1362
+ // inverse 2.682 admitted vs direct 1.468 refused). Reading the IDF sum
1363
+ // makes the verdict mode-independent, and changes nothing in the
1364
+ // engine's own mode, where the two readings coincide.
1365
+ passesNaturalBreak = point.idfVote >= rootCut;
1366
+ passesConsensusFloor = point.idfVote >= floor;
1234
1367
  // CO-DOMINANT — an anchor the estimator cannot separate from the
1235
1368
  // dominant inherits the dominant's exemption, because that exemption's
1236
1369
  // only warrant is being TOP, and "top" is not a fact about the corpus
@@ -1685,6 +1818,13 @@ ownRootsA, ownRootsB, trace) {
1685
1818
  outcome,
1686
1819
  });
1687
1820
  };
1821
+ // ── ADMISSION BY MARGIN, not by votes (see voteRegions' note) ─────────
1822
+ // What this site measures: how far the best ANN proposal's effective score
1823
+ // (score × semanticConfidence) stands above the runner-up's, against
1824
+ // `estimatorNoise(D)`. What it does NOT measure: how many regions voted,
1825
+ // or whether the query's regions agree — that is voteRegions' question, and
1826
+ // here a synthetic gist has already replaced them. The two bars are both
1827
+ // derived (thresholds.md), and neither is a tuning of the other.
1688
1828
  let selected = null;
1689
1829
  let selectedReach = null;
1690
1830
  let selectedIdf = 0;
@@ -1790,6 +1930,15 @@ async function crossRegionVotes(ctx, query, regions, rvs, k, N, reachMemo, td) {
1790
1930
  // successfully reconstructed while probing one pair must not be read and
1791
1931
  // perceived again while probing another pair in the same climb.
1792
1932
  const siblingGistMemo = new Map();
1933
+ // ── ADMISSION BY STRUCTURE, not by a bar (see voteRegions' note) ──────
1934
+ // What this site decides: WHICH regions may pair at all — a region that
1935
+ // already voted (individually discriminative), or a KNOWN non-voting one as
1936
+ // the weak side of a pair whose other side voted; never two non-voting
1937
+ // regions, and never a span contained in a maximal one whose reading is
1938
+ // exact. The bar (the container's idf) comes later, on the candidate. So
1939
+ // its "rejection reasons" name structural disqualifications — a different
1940
+ // vocabulary because it answers a different question, and the three
1941
+ // taxonomies stay separate for the same reason the readings do.
1793
1942
  const votedSpans = new Set();
1794
1943
  for (const rv of rvs.votes)
1795
1944
  votedSpans.add(`${rv.start},${rv.end}`);
@@ -23,6 +23,22 @@ export declare function leafIdRun(ctx: MindContext, bytes: Uint8Array, from: num
23
23
  * what a partial prefix means (deposit only interns the whole-stream flat
24
24
  * branch when the prefix covers everything). */
25
25
  export declare function leafIdPrefix(ctx: MindContext, bytes: Uint8Array): number[];
26
+ /** Which prefixes of `prefix ‖ tail` are STORED NODES — as lengths in the
27
+ * tail's own coordinates, ascending, excluding the empty one. This is the
28
+ * candidate set a rule needs to join an already-stored prefix to a suffix it
29
+ * has not stored: a key names a relation exactly when `prefix ‖ tail[0..p]` IS
30
+ * a node, and a key can end strictly inside the tail without sitting on any
31
+ * fold boundary (a stored member's end is the end of ITS OWN stream, and the
32
+ * fold never emits a cut at a stream's end). Measured: "stockholm mayor"
33
+ * exists, leads on, and its boundary 6 is in neither the tail's cuts nor the
34
+ * concatenation's.
35
+ *
36
+ * ONE cheap content-addressed probe per offset — `leafIdPrefix` walks the bytes
37
+ * once (a point probe each), `findBranch` hashes the growing kid run — and NO
38
+ * `resolve`, which is what keeps this off the O(suffix) vector folds the
39
+ * recognition path pays. It stops at the first byte that was never interned,
40
+ * which costs nothing real: a stored key's bytes are interned by construction. */
41
+ export declare function keyEnds(ctx: MindContext, prefix: Uint8Array, tail: Uint8Array): number[];
26
42
  /** The canonical W-window node ids of a byte stream, offset → id — the
27
43
  * CONTENT-ADDRESSED IDENTITY of every W-sized slice, under which any content
28
44
  * two deposits share IS the same node (hash-consing paid the comparison at
@@ -70,6 +70,47 @@ export function leafIdPrefix(ctx, bytes) {
70
70
  }
71
71
  return ids;
72
72
  }
73
+ /** Which prefixes of `prefix ‖ tail` are STORED NODES — as lengths in the
74
+ * tail's own coordinates, ascending, excluding the empty one. This is the
75
+ * candidate set a rule needs to join an already-stored prefix to a suffix it
76
+ * has not stored: a key names a relation exactly when `prefix ‖ tail[0..p]` IS
77
+ * a node, and a key can end strictly inside the tail without sitting on any
78
+ * fold boundary (a stored member's end is the end of ITS OWN stream, and the
79
+ * fold never emits a cut at a stream's end). Measured: "stockholm mayor"
80
+ * exists, leads on, and its boundary 6 is in neither the tail's cuts nor the
81
+ * concatenation's.
82
+ *
83
+ * ONE cheap content-addressed probe per offset — `leafIdPrefix` walks the bytes
84
+ * once (a point probe each), `findBranch` hashes the growing kid run — and NO
85
+ * `resolve`, which is what keeps this off the O(suffix) vector folds the
86
+ * recognition path pays. It stops at the first byte that was never interned,
87
+ * which costs nothing real: a stored key's bytes are interned by construction. */
88
+ export function keyEnds(ctx, prefix, tail) {
89
+ if (prefix.length === 0 || tail.length === 0)
90
+ return [];
91
+ const joined = new Uint8Array(prefix.length + tail.length);
92
+ joined.set(prefix, 0);
93
+ joined.set(tail, prefix.length);
94
+ const ids = leafIdPrefix(ctx, joined);
95
+ if (ids.length < prefix.length)
96
+ return [];
97
+ const ends = [];
98
+ // The kid run GROWS by one id per offset; `findBranch` wants an array, so the
99
+ // run is built once and pushed into, never re-sliced. Re-slicing
100
+ // `ids.slice(0, prefix.length + p)` per offset made this O(|tail| ·
101
+ // (|prefix| + |tail|)) — quadratic in the tail, where the learning path this
102
+ // follows slices a run that SHRINKS. Same ends, linear copying.
103
+ const run = ids.slice(0, prefix.length);
104
+ // The loop ENDS at the first byte that was never interned (`ids.length`):
105
+ // every later prefix contains it, so none of them can be a node either — this
106
+ // is where the scan stops, not a silent truncation of the answer.
107
+ for (let p = 1; prefix.length + p <= ids.length; p++) {
108
+ run.push(ids[prefix.length + p - 1]);
109
+ if (ctx.store.findBranch(run) !== null)
110
+ ends.push(p);
111
+ }
112
+ return ends;
113
+ }
73
114
  /** The canonical W-window node ids of a byte stream, offset → id — the
74
115
  * CONTENT-ADDRESSED IDENTITY of every W-sized slice, under which any content
75
116
  * two deposits share IS the same node (hash-consing paid the comparison at
@@ -0,0 +1,40 @@
1
+ import type { MindContext } from "./types.js";
2
+ /** One stored experience pair, as bytes. */
3
+ export interface CorpusPair {
4
+ context: Uint8Array;
5
+ continuation: Uint8Array;
6
+ contextId: number;
7
+ continuationId: number;
8
+ /** Bytes of the query this pair was matched on — 0 when browsing. */
9
+ matchedBytes: number;
10
+ /** True when the stored bytes ran past the declared preview capacity. */
11
+ contextTruncated: boolean;
12
+ continuationTruncated: boolean;
13
+ }
14
+ /** Why a search produced no pairs. A STATE, so a caller's own layer can say it
15
+ * in its own words — the byte layer does not speak. */
16
+ export type CorpusMiss = "matched" | "nothing-resolved" | "no-continuations";
17
+ export interface CorpusResult {
18
+ pairs: CorpusPair[];
19
+ /** Query subtrees that content-addressed to a real stored node. */
20
+ resolved: number;
21
+ /** Distinct edge-bearing contexts the climb reached. */
22
+ reached: number;
23
+ /** Distinct contexts that carry a learnt continuation, store-wide. */
24
+ totalContexts: number;
25
+ /** True when these are browse samples rather than search results. */
26
+ browsed: boolean;
27
+ miss: CorpusMiss;
28
+ }
29
+ /** Which stored notes does this query reach? BYTES in, BYTES out.
30
+ *
31
+ * Exact content addressing through the machinery that already exists: the
32
+ * query's recognised sites are the resolved subtrees, the climb goes up from
33
+ * the biggest first, and a pair is a context that carries a continuation. */
34
+ export declare function searchCorpus(ctx: MindContext, queryBytes: Uint8Array, limit?: number): CorpusResult;
35
+ /** Browse real pairs, striding the id space so the sample is spread rather than
36
+ * one local cluster. DETERMINISTIC: `from` is the caller's own offset, so
37
+ * browsing twice with different offsets shows different notes without a random
38
+ * draw (the demo drew `Math.random()`, which the engine cannot do — same seed,
39
+ * same order, same query must mean the same answer). */
40
+ export declare function sampleCorpus(ctx: MindContext, limit?: number, from?: number): CorpusResult;
@@ -0,0 +1,149 @@
1
+ // corpus.ts — read the trained memory back out of the DAG, AS DATA.
2
+ //
3
+ // A trained experience pair IS one continuation edge: `src` is the context that
4
+ // was deposited, `dst` is what the mind learnt follows it. Reading them back
5
+ // uses the store's own structure and its own indexes — no auxiliary index is
6
+ // built, nothing is written, and NOTHING HERE KNOWS ABOUT TEXT: this layer takes
7
+ // bytes and returns bytes. The text case is one helper on the Mind
8
+ // (`searchCorpusText`), which encodes, calls this, and decodes.
9
+ //
10
+ // WHERE EACH STAGE COMES FROM (ported from the demo's `explore.ts`, which
11
+ // hand-rolled its own resolution):
12
+ //
13
+ // 1. PERCEIVE, CONTENT-ADDRESS and ADMIT the query — `recognise()`, the SAME
14
+ // machinery an answer goes through. It returns the sites: the query spans
15
+ // that content-addressed to a stored node that can lead somewhere. The
16
+ // demo's own recursive `findLeaf`/`findBranch` walk was a second
17
+ // implementation of exactly this, and it is NOT ported.
18
+ // 2. CLIMB the structural `kid` table from each resolved site to the
19
+ // edge-bearing contexts above it (`edgeAncestors`, traverse.ts), weighting
20
+ // each context by how much query content reached it.
21
+ // 3. READ the continuation off the edge table (`nextFirst`).
22
+ //
23
+ // COST is set by how much of the QUERY resolves, never by the size of the
24
+ // store: the sites are what recognition already found, the climb is bounded by
25
+ // the declared `corpusClimbs`/`corpusContextsPerClimb`, and every store call is
26
+ // a point probe or a capped read. All work is accounted by the store's own
27
+ // meter hooks when a response's meter is open — there is no second instrument.
28
+ //
29
+ // WHAT THIS IS NOT. Exact content addressing, not fuzzy keyword search: a query
30
+ // shares results with a stored note when it shares actual chunk-aligned content
31
+ // with it. An arbitrary mid-word fragment resolves to nothing, and the honest
32
+ // answer there is "nothing matched" — which is why `sampleCorpus` exists, and
33
+ // why the miss is reported as a STATE rather than as prose (the text helper
34
+ // turns it into words).
35
+ import { recognise } from "./recognition.js";
36
+ import { edgeAncestors } from "./traverse.js";
37
+ /** A context node as a pair, or null when it carries no continuation. */
38
+ function pairOf(ctx, id, matchedBytes) {
39
+ const outs = ctx.store.nextFirst(id, 1);
40
+ if (outs.length === 0)
41
+ return null;
42
+ const cap = ctx.cfg.corpusPreviewBytes;
43
+ const context = ctx.store.bytesPrefix(id, cap + 1);
44
+ const continuation = ctx.store.bytesPrefix(outs[0], cap + 1);
45
+ const contextTruncated = context.length > cap;
46
+ const continuationTruncated = continuation.length > cap;
47
+ return {
48
+ context: contextTruncated ? context.subarray(0, cap) : context,
49
+ continuation: continuationTruncated
50
+ ? continuation.subarray(0, cap)
51
+ : continuation,
52
+ contextId: id,
53
+ continuationId: outs[0],
54
+ matchedBytes,
55
+ contextTruncated,
56
+ continuationTruncated,
57
+ };
58
+ }
59
+ /** Which stored notes does this query reach? BYTES in, BYTES out.
60
+ *
61
+ * Exact content addressing through the machinery that already exists: the
62
+ * query's recognised sites are the resolved subtrees, the climb goes up from
63
+ * the biggest first, and a pair is a context that carries a continuation. */
64
+ export function searchCorpus(ctx, queryBytes, limit) {
65
+ const store = ctx.store;
66
+ const want = Math.max(1, Math.min(limit ?? ctx.cfg.corpusContextsPerClimb, ctx.cfg.corpusLimitMax));
67
+ // A resolved subtree must account for at least one window (W): a single
68
+ // character resolves against almost any store and means nothing. W is the
69
+ // mind's own line between chance and evidence — derived, never declared.
70
+ const floor = ctx.space.maxGroup;
71
+ const resolved = recognise(ctx, queryBytes).sites
72
+ .map((s) => ({ id: s.payload, len: store.contentLen(s.payload, 512) }))
73
+ .filter((r) => r.len >= floor);
74
+ // Biggest first, lowest id breaking ties: a clause is evidence, a character is
75
+ // noise, and equal evidence must not be decided by iteration order.
76
+ const byLength = resolved
77
+ .sort((a, b) => b.len - a.len || a.id - b.id)
78
+ .slice(0, ctx.cfg.corpusClimbs);
79
+ // Weight each context by how much query content reached it.
80
+ const weight = new Map();
81
+ for (const { id, len } of byLength) {
82
+ for (const root of edgeAncestors(ctx, id, ctx.cfg.corpusContextsPerClimb).roots) {
83
+ weight.set(root, (weight.get(root) ?? 0) + len);
84
+ }
85
+ }
86
+ const pairs = [];
87
+ const ranked = [...weight.entries()].sort((a, b) => b[1] - a[1] || a[0] - b[0]);
88
+ for (const [id, w] of ranked) {
89
+ if (pairs.length >= want)
90
+ break;
91
+ const pair = pairOf(ctx, id, w);
92
+ if (pair)
93
+ pairs.push(pair);
94
+ }
95
+ return {
96
+ pairs,
97
+ resolved: byLength.length,
98
+ reached: weight.size,
99
+ totalContexts: store.edgeSourceCount(),
100
+ browsed: false,
101
+ miss: pairs.length > 0
102
+ ? "matched"
103
+ : weight.size === 0
104
+ ? "nothing-resolved"
105
+ : "no-continuations",
106
+ };
107
+ }
108
+ /** Browse real pairs, striding the id space so the sample is spread rather than
109
+ * one local cluster. DETERMINISTIC: `from` is the caller's own offset, so
110
+ * browsing twice with different offsets shows different notes without a random
111
+ * draw (the demo drew `Math.random()`, which the engine cannot do — same seed,
112
+ * same order, same query must mean the same answer). */
113
+ export function sampleCorpus(ctx, limit, from = 0) {
114
+ const store = ctx.store;
115
+ const want = Math.max(1, Math.min(limit ?? ctx.cfg.corpusContextsPerClimb, ctx.cfg.corpusLimitMax));
116
+ const total = store.nodeCount();
117
+ const probes = ctx.cfg.corpusSampleProbes;
118
+ const floorBytes = ctx.cfg.corpusSampleFloorBytes;
119
+ const pairs = [];
120
+ // EACH CONTEXT AT MOST ONCE. Striding the id space revisits ids when the
121
+ // store is small relative to the probe budget (measured: a 160-node store
122
+ // returned the SAME pair six times for `limit: 6`), and a browse that repeats
123
+ // itself is not a browse. The demo had the same hole; it is invisible only on
124
+ // a store far larger than the probe budget.
125
+ const seen = new Set();
126
+ for (let i = 0; i < probes && pairs.length < want && total > 0; i++) {
127
+ const slot = (i / probes + from) % 1;
128
+ const id = Math.floor(slot * total);
129
+ if (seen.has(id))
130
+ continue;
131
+ if (!store.has(id) || !store.hasNext(id))
132
+ continue;
133
+ if (store.contentLen(id, floorBytes) < floorBytes)
134
+ continue;
135
+ const pair = pairOf(ctx, id, 0);
136
+ if (pair) {
137
+ seen.add(id);
138
+ pairs.push(pair);
139
+ }
140
+ }
141
+ return {
142
+ pairs,
143
+ resolved: 0,
144
+ reached: pairs.length,
145
+ totalContexts: store.edgeSourceCount(),
146
+ browsed: true,
147
+ miss: pairs.length > 0 ? "matched" : "no-continuations",
148
+ };
149
+ }
@@ -162,6 +162,13 @@ export declare class GraphSearch {
162
162
  * recursive completion), and chooseNext (distributional-evidence edge
163
163
  * disambiguation when a recognised form has multiple continuations). */
164
164
  host: GraphSearchHost);
165
+ /** The nodes the QUERY canonically names — the same identity the store's keys
166
+ * were written through. A byte-exact test is not enough: the query writes
167
+ * `Eiffel Tower country` and the deposited node is `eiffel tower country`, so
168
+ * a join that filters the query's own subject by RAW bytes re-admits it —
169
+ * measured: that is the trap's wrong answer (`The capital of Eiffel Tower
170
+ * country is Berlin.`). Cached by query identity, because the search is
171
+ * reused across responses. */
165
172
  private hubBound;
166
173
  /** Explore the Sema graph for the lightest cover of the query and return its
167
174
  * chosen spans left-to-right — WITH the derivation's total weight (the g