@hviana/sema 0.8.1 → 0.8.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (114) hide show
  1. package/AGENTS.md +29 -29
  2. package/TRADEMARKS.md +0 -1
  3. package/dist/src/config.d.ts +28 -0
  4. package/dist/src/config.js +20 -0
  5. package/dist/src/geometry.d.ts +21 -0
  6. package/dist/src/geometry.js +21 -0
  7. package/dist/src/meter.d.ts +76 -0
  8. package/dist/src/meter.js +95 -0
  9. package/dist/src/mind/attention.d.ts +4 -0
  10. package/dist/src/mind/attention.js +165 -16
  11. package/dist/src/mind/canonical.d.ts +16 -0
  12. package/dist/src/mind/canonical.js +41 -0
  13. package/dist/src/mind/corpus.d.ts +40 -0
  14. package/dist/src/mind/corpus.js +149 -0
  15. package/dist/src/mind/graph-search.d.ts +7 -0
  16. package/dist/src/mind/graph-search.js +254 -24
  17. package/dist/src/mind/index.d.ts +3 -1
  18. package/dist/src/mind/index.js +1 -0
  19. package/dist/src/mind/match.d.ts +9 -4
  20. package/dist/src/mind/match.js +147 -61
  21. package/dist/src/mind/mechanisms/cast.js +19 -3
  22. package/dist/src/mind/mechanisms/confluence.js +24 -0
  23. package/dist/src/mind/mechanisms/cover.js +6 -0
  24. package/dist/src/mind/mechanisms/recall.js +32 -4
  25. package/dist/src/mind/mind.d.ts +57 -0
  26. package/dist/src/mind/mind.js +72 -1
  27. package/dist/src/mind/pipeline-mechanism.d.ts +7 -0
  28. package/dist/src/mind/pipeline.js +66 -20
  29. package/dist/src/mind/primitives.js +9 -1
  30. package/dist/src/mind/rationale.d.ts +28 -1
  31. package/dist/src/mind/rationale.js +22 -1
  32. package/dist/src/mind/reasoning.d.ts +25 -3
  33. package/dist/src/mind/reasoning.js +125 -20
  34. package/dist/src/mind/recognition.js +4 -8
  35. package/dist/src/mind/resonance.js +20 -1
  36. package/dist/src/mind/trace.js +1 -0
  37. package/dist/src/mind/traverse.js +15 -3
  38. package/dist/src/mind/types.d.ts +49 -4
  39. package/docs/INVARIANTS.md +2 -2
  40. package/docs/architecture/bounded-reads.md +1 -1
  41. package/docs/architecture/commonality.md +2 -2
  42. package/docs/architecture/cost-model.md +2 -2
  43. package/docs/architecture/determinism.md +7 -7
  44. package/docs/architecture/match-project.md +2 -3
  45. package/docs/architecture/mechanism-market.md +10 -10
  46. package/docs/architecture/meter.md +5 -5
  47. package/docs/architecture/store.md +3 -3
  48. package/docs/failures/tempting-but-wrong.md +34 -6
  49. package/docs/harness/gates.md +2 -2
  50. package/docs/mechanisms/cast.md +2 -2
  51. package/docs/mechanisms/cover.md +2 -3
  52. package/docs/mechanisms/extraction.md +7 -7
  53. package/docs/mechanisms/recall.md +8 -9
  54. package/jsr.json +1 -1
  55. package/package.json +1 -1
  56. package/src/alu/README.md +11 -12
  57. package/src/config.ts +48 -0
  58. package/src/geometry.ts +21 -0
  59. package/src/meter.ts +98 -0
  60. package/src/mind/attention.ts +167 -16
  61. package/src/mind/canonical.ts +43 -0
  62. package/src/mind/corpus.ts +202 -0
  63. package/src/mind/graph-search.ts +277 -23
  64. package/src/mind/index.ts +8 -1
  65. package/src/mind/match.ts +148 -57
  66. package/src/mind/mechanisms/cast.ts +20 -2
  67. package/src/mind/mechanisms/confluence.ts +24 -0
  68. package/src/mind/mechanisms/cover.ts +5 -0
  69. package/src/mind/mechanisms/recall.ts +32 -4
  70. package/src/mind/mind.ts +125 -0
  71. package/src/mind/pipeline-mechanism.ts +7 -0
  72. package/src/mind/pipeline.ts +79 -22
  73. package/src/mind/primitives.ts +9 -1
  74. package/src/mind/rationale.ts +35 -1
  75. package/src/mind/reasoning.ts +145 -13
  76. package/src/mind/recognition.ts +4 -8
  77. package/src/mind/resonance.ts +19 -1
  78. package/src/mind/trace.ts +1 -0
  79. package/src/mind/traverse.ts +16 -6
  80. package/src/mind/types.ts +53 -4
  81. package/test/100-complete-grounding-trace.test.mjs +109 -0
  82. package/test/101-alignment-gap-bound.test.mjs +106 -0
  83. package/test/102-production-composes-at-scale.test.mjs +110 -0
  84. package/test/103-alignment-gap-budget.test.mjs +89 -0
  85. package/test/104-composition-is-reported.test.mjs +90 -0
  86. package/test/105-derive-through-reports-its-refusal.test.mjs +137 -0
  87. package/test/106-the-join-fires.test.mjs +94 -0
  88. package/test/107-the-join-is-counted.test.mjs +81 -0
  89. package/test/108-the-join-chains.test.mjs +78 -0
  90. package/test/109-the-pivot-is-counted.test.mjs +60 -0
  91. package/test/110-the-reasoner-stops-when-the-question-is-answered.test.mjs +91 -0
  92. package/test/111-the-cover-assembly-is-counted.test.mjs +74 -0
  93. package/test/112-the-exploration-does-not-grow-with-the-hub.test.mjs +89 -0
  94. package/test/113-the-rationale-payload-is-bounded.test.mjs +84 -0
  95. package/test/114-alignment-budget-is-per-sweep.test.mjs +93 -0
  96. package/test/116-the-extension-is-gated-by-the-pipelines-own-remainder.test.mjs +100 -0
  97. package/test/117-corpus-search.test.mjs +171 -0
  98. package/test/118-the-join-reaches-a-key-off-the-cut.test.mjs +74 -0
  99. package/test/119-the-work-does-not-grow-with-the-corpus.test.mjs +122 -0
  100. package/test/120-composition-is-consequence.test.mjs +132 -0
  101. package/test/121-the-extension-does-not-grow-with-the-corpus.test.mjs +128 -0
  102. package/test/122-the-climb-search-does-not-grow-with-the-corpus.test.mjs +117 -0
  103. package/test/123-the-paired-formulas-agree.test.mjs +90 -0
  104. package/test/125-the-post-grounding-branch-publishes-its-operand.test.mjs +51 -0
  105. package/test/126-the-pipeline-does-not-name-mechanisms.test.mjs +42 -0
  106. package/test/128-the-leads-somewhere-pair-agrees.test.mjs +83 -0
  107. package/test/129-the-trace-payload-shape.test.mjs +164 -0
  108. package/test/14-scaling.test.mjs +10 -7
  109. package/test/32-confluence.test.mjs +68 -0
  110. package/test/38-reason-restate-guard.test.mjs +8 -2
  111. package/test/43-cast-analog-seat.test.mjs +10 -0
  112. package/test/55-cost-meter.test.mjs +859 -0
  113. package/test/76-reference-binding.test.mjs +6 -1
  114. package/test/89-completion-recursion.test.mjs +30 -5
package/src/geometry.ts CHANGED
@@ -155,6 +155,27 @@ export function profileCapacity(D: number): number {
155
155
  return Math.max(1, Math.floor(Math.sqrt(D)));
156
156
  }
157
157
 
158
+ /**
159
+ * The POOLED-vote significance floor, and the derivation lives here because
160
+ * its PREMISE is a property of the caller's weighting.
161
+ *
162
+ * DERIVATION (docs/architecture/thresholds.md §2): a maximally-specific region
163
+ * contributes at most `ln N` to a pooled vote, so `ln(N) + 1/2` sits half a
164
+ * unit above ONE region's ceiling — it demands corroboration BEYOND a single
165
+ * region, which is what makes it a consensus bar rather than a resonance bar.
166
+ *
167
+ * PREMISE: that per-region ceiling is an IDF, `ln(N/c)` — attention.ts's
168
+ * `inverse` mode, the mode every non-test caller runs. The other two modes
169
+ * weight a region by `ln(1+c)` (`direct`) or `ln(N/c) + ln(1+c)` (`combined`),
170
+ * i.e. `ln N + ln(1 + 1/c)`, so they exceed the premise's ceiling by at most
171
+ * `ln 2` — a DERIVED bound, not a hole: the floor stays within `ln 2` of its
172
+ * own premise in every mode, and exactly on it in `inverse`.
173
+ *
174
+ * MEASURED: the floor is read on the pooled vote (`commitVotes`, `recall`,
175
+ * `cast`). Across 27 anchors on 6 queries, 11 cleared it by the sum and NONE
176
+ * by a single region's peak — gating on one region would refuse every elected
177
+ * root.
178
+ */
158
179
  export function consensusFloor(N: number): number {
159
180
  return Math.log(N) + 1 / 2;
160
181
  }
package/src/meter.ts CHANGED
@@ -206,6 +206,104 @@ export class Meter {
206
206
  /** Candidates the decider weighed. */
207
207
  candidates = 0;
208
208
 
209
+ // ── Graph search: the fact join (DIRECTION) ─────────────────────────────
210
+ //
211
+ // The join's outcome was observable ONLY through the rationale, and the
212
+ // rationale PERTURBS the search (measured: appending text to a refusal note
213
+ // changed a traced answer). These four counters are the untraced view — the
214
+ // same surface every other work counter uses, incremented where the decision
215
+ // is made, never behind a trace guard.
216
+ /** `deriveThrough` yielded — a fact was reached through the subject the query
217
+ * never named. */
218
+ joinFired = 0;
219
+ /** Refused: no key names the entity and the tail together. (A key that
220
+ * resolves but leads nowhere is not "refused" — it is not the relation, so
221
+ * the scan simply moves on; there is no counter for a case the loop cannot
222
+ * reach.) */
223
+ joinNoKey = 0;
224
+ /** Refused: the fact contains no entity that leads anywhere. */
225
+ joinNoEntity = 0;
226
+ /** `recompleteNode` re-covered a produced form — the descent that decomposes
227
+ * a completion by ITS OWN kids. Without this the descent is invisible: a
228
+ * caller could see the chain's result but not whether the recomposition
229
+ * happened, so "the recursion stopped" and "the recursion never ran" were
230
+ * indistinguishable from the counters alone. */
231
+ recompletes = 0;
232
+
233
+ // ── Mind: the multi-hop pivot (EXTENSION) ───────────────────────────────
234
+ //
235
+ // `pivotStep` was observable only through the rationale, and the rationale
236
+ // perturbs the search (measured). How far the reasoner hopped is a
237
+ // BEHAVIOUR, so it needs an untraced view: one counter, incremented where the
238
+ // step is emitted.
239
+ /** Times the reasoner pivoted on a span its answer contains and stepped
240
+ * across that fact. */
241
+ pivotSteps = 0;
242
+ /** Canon probes REFUSED because the canon budget ran out — the one thing the
243
+ * budget does that nothing could see. The budget itself is derived
244
+ * (`bytes.length · chainReach(W)²`, recognition.ts), and the cheap exact route
245
+ * is deliberately unbudgeted, so this counter says exactly when the expensive
246
+ * route was priced out. Counted where the fact happens (the `!canonBudget`
247
+ * refusal), not where the probe is called. */
248
+ canonProbesDenied = 0;
249
+ /** The pipeline's remainder AT THE DECISION POINT, in bytes: what the grounded
250
+ * answer plus the pre-computed spans left unexplained, after the same W floor
251
+ * the fuse gate uses. This is the quantity that licenses (or refuses) the
252
+ * post-grounding extension and the fusion — it was computed, used, and never
253
+ * published, so nothing could measure what a search had LEFT when it decided.
254
+ * Read with {@link postGroundingRemainderSpans}. */
255
+ postGroundingRemainderBytes = 0;
256
+ /** How many spans that remainder consists of (each at least one W window). */
257
+ postGroundingRemainderSpans = 0;
258
+ /** Times `fuseAttention` produced a FUSED answer — not times it was called.
259
+ * It is entered whenever the query has a remainder ≥ W and returns early when
260
+ * there is nothing to bridge (`containsSpan`, a lone root, an empty pass), so
261
+ * the call and the fact are different things and only the fact is counted.
262
+ * Its own rationale step reports the fusion; this is the untraced view, and
263
+ * its cost is one bridging edge: `fuseRuns · STEP`. */
264
+ fuseRuns = 0;
265
+ /** Steps the post-grounding EXTENSION took — pivots plus forward-absorbs.
266
+ * `pivotSteps` counts only the former, so before this the extension's COST was
267
+ * not computable at all. With it, the price of extending the answer is
268
+ * `reasonSteps · STEP`, the ladder's own value for following an edge. */
269
+ reasonSteps = 0;
270
+ /** Bytes of the grounding's UNCOVERED material the extension was justified by
271
+ * — the union of the spans each step carried a `W`-window of. The gate
272
+ * already computed WHICH span carried it per step and kept only a boolean;
273
+ * this is that fact, accumulated. Read with {@link reasonSteps}: one is the
274
+ * price, the other the explanation. */
275
+ reasonCarriedBytes = 0;
276
+ /** Branch-node probes the pivot sweep actually spent looking for the learnt
277
+ * context an answer contains (one `resonate` per probe). The untraced view
278
+ * of what the multi-hop's shortlist costs. */
279
+ pivotProbes = 0;
280
+ /** Branch nodes the pivot's probe cap withheld (`branchCount − probeCap`, over
281
+ * every call). A capacity fact, not a verdict: the sweep is breadth-first,
282
+ * so the probes it DOES spend are the largest regions, and recognition still
283
+ * contributes every exact containment candidate regardless of the budget.
284
+ * Read it with {@link pivotProbes} — one says the work, the other the
285
+ * shortfall. */
286
+ pivotBranchesUnprobed = 0;
287
+
288
+ // ── Mind: the cover's connector assembly (LIMIT) ────────────────────────
289
+ //
290
+ // The cover's `run` is 91% of a hub query's time (`"Hello."`: 2.7 s of 3.0 s)
291
+ // and holds its ~270 MB peak, and none of it was countable: `searchPushes`
292
+ // and `candidates` do not see the connector assembly. These two counters are
293
+ // the untraced view of it.
294
+ /** `bridge` calls the cover makes assembling connectors (pairwise + n-ary). */
295
+ coverBridges = 0;
296
+ /** Continuations a CHAIN hop offered the search. Bounded by the question
297
+ * (`ceil(queryLen / W)`) rather than by the corpus's fan-out — measured on a
298
+ * hub of degree 1083, offering every continuation grew the chart to 3113 outs
299
+ * and cost a 270 MB peak / 256 MB OOM for a two-word question. */
300
+ chainOffers = 0;
301
+ /** Σ byte-allowance the n-ary interior passes those bridges. The allowance
302
+ * is `middleBytes + (m + 1) * W` — every intermediate answer's bytes plus
303
+ * one window of glue per joint — so it is the quantity that grows with a hub
304
+ * query's answers, and the first thing to read when the peak moves. */
305
+ coverAllowanceBytes = 0;
306
+
209
307
  // ── Phases ──────────────────────────────────────────────────────────────
210
308
 
211
309
  private readonly _phases = new Map<string, PhaseCost>();
@@ -180,6 +180,10 @@ export interface ConsensusAnchorTrace {
180
180
 
181
181
  pooledVote: number;
182
182
  idfVote: number;
183
+ /** The LARGEST single-region contribution behind this anchor — the bar
184
+ * recall's own gate reads (mechanisms/recall.ts). Published so the one
185
+ * decision-making quantity the climb computes is not invisible. */
186
+ peak: number;
183
187
 
184
188
  candidateBreadth: number;
185
189
  contributingVotes: number;
@@ -1243,15 +1247,18 @@ export async function voteRegions(
1243
1247
  }
1244
1248
  contrastiveMargin = margin;
1245
1249
  // Scaled by what this region does NOT address — see `cov` above.
1246
- const noiseFloor = estimatorNoise(ctx.store.D) * (1 - cov);
1247
- if (margin <= noiseFloor) {
1250
+ // The bar THIS gate applies: the estimator's noise scaled by what the
1251
+ // region does NOT address (`cov`). ONE definition, used by the rejection
1252
+ // path below and by the voted payload — the trace reports the applied bar.
1253
+ const appliedFloor = estimatorNoise(ctx.store.D) * (1 - cov);
1254
+ if (margin <= appliedFloor) {
1248
1255
  recordRegion("contrastive-margin-rejection", {
1249
1256
  selected,
1250
1257
  reachNode: voterId,
1251
1258
  idf,
1252
1259
  dfWeight: wf,
1253
1260
  contrastiveMargin: margin,
1254
- contrastiveNoiseFloor: noiseFloor,
1261
+ contrastiveNoiseFloor: appliedFloor,
1255
1262
  ...(contrastiveRival ? { contrastiveRival } : {}),
1256
1263
  });
1257
1264
  continue;
@@ -1305,7 +1312,12 @@ export async function voteRegions(
1305
1312
  ...(contrastiveMargin !== undefined
1306
1313
  ? {
1307
1314
  contrastiveMargin,
1308
- contrastiveNoiseFloor: estimatorNoise(ctx.store.D),
1315
+ // THE BAR THE GATE ACTUALLY APPLIED — the same expression
1316
+ // the rejection path's `appliedFloor` defines, inline here because
1317
+ // this payload is built in a scope that does not carry that local.
1318
+ // Publishing the raw estimatorNoise(D) instead made a region that
1319
+ // PASSED look closer to its limit than it was.
1320
+ contrastiveNoiseFloor: estimatorNoise(ctx.store.D) * (1 - cov),
1309
1321
  ...(contrastiveRival ? { contrastiveRival } : {}),
1310
1322
  }
1311
1323
  : {}),
@@ -1438,7 +1450,15 @@ export function poolVotes(
1438
1450
  },
1439
1451
  pool,
1440
1452
  };
1441
- lightestDerivation(system);
1453
+ // THE SEARCH WAS THE ONE LAYER WITH NO TIME. The climb's phases are timed
1454
+ // (voteRegions, structuralResonance, crossRegion) but the pooled derivation
1455
+ // was not, so any cost or gain inside it stayed invisible.
1456
+ // `timeSync`, not `time`: the search is SYNCHRONOUS, and wrapping it in a
1457
+ // promise only to time it would make the profiled path wait where the
1458
+ // unprofiled one does not (meter.ts's own contract).
1459
+ if (ctx.meter) {
1460
+ ctx.meter.timeSync("climb.derivation", () => lightestDerivation(system));
1461
+ } else lightestDerivation(system);
1442
1462
 
1443
1463
  const votes = new Map<number, number>();
1444
1464
  const votesIdf = new Map<number, number>();
@@ -1476,12 +1496,24 @@ export function poolVotes(
1476
1496
  // The LARGEST single region's contribution to this anchor's pooled vote.
1477
1497
  // The pool is a SUM (deliberately — see the pooling note above), so it says
1478
1498
  // how much evidence there is in total, never whether any ONE place in the
1479
- // query carries evidence on its own. Consumers that hold an anchor to
1480
- // consensusFloor(N) = ln(N) + 1/2 need the latter: that bar prices ONE
1481
- // region's maximally-discriminative evidence (ln N is the IDF of content
1482
- // reaching a single context), so comparing a six-region sum against it is a
1483
- // dimensional error. Recorded here, beside the count, because this is the
1484
- // only place the per-region contributions are still separable.
1499
+ // query carries evidence on its own. Recorded here, beside the count,
1500
+ // because this is the only place the per-region contributions are still
1501
+ // separable.
1502
+ //
1503
+ // THE BAR IS THE POOLED FLOOR, AND IT WAS ONCE CLAIMED OTHERWISE HERE.
1504
+ // This comment used to say that holding an anchor to consensusFloor(N)
1505
+ // "prices ONE region's evidence", so comparing a six-region sum against it
1506
+ // was "a dimensional error". THAT WAS FALSE. `thresholds.md` §2 derives
1507
+ // `consensusFloor` as the POOLED-vote significance floor ("each region
1508
+ // contributes at most ln(N/c) <= ln(N); ln(N) + 1/2 demands ..."), and the
1509
+ // climb weights by IDF, so the sum and the floor are in ONE dimension —
1510
+ // which is exactly why `recall.ts` gates `forest[0].idfVote` against it and
1511
+ // why `commitVotes` does too. The other two weighting modes DO leave that
1512
+ // dimension (by at most ln 2, two-sided: `direct` deflates a region and
1513
+ // `combined` inflates it), and the gates therefore read the IDF sum, which
1514
+ // is mode-independent; `test/55` tests 19 and 20 pin both halves — the sum
1515
+ // as the reading the bar is derived for, and the absence of any gate
1516
+ // inversion across the three modes.
1485
1517
  const regionPeak = new Map<number, number>();
1486
1518
  const steps: DerivationStep[] = [];
1487
1519
  let order = 0;
@@ -1665,6 +1697,7 @@ export function commitVotes(
1665
1697
  anchor,
1666
1698
  vote,
1667
1699
  peak: regionPeak.get(anchor) ?? 0,
1700
+ idfVote: votesIdf.get(anchor) ?? 0,
1668
1701
  start: s.start,
1669
1702
  end: s.end,
1670
1703
  breadth: (regionSupport.get(anchor) ?? 0) / totalRegions,
@@ -1674,6 +1707,16 @@ export function commitVotes(
1674
1707
  ),
1675
1708
  };
1676
1709
  })
1710
+ // THE ORDER IS NOT A PREFERENCE: with equal evidence it decides ADMISSION,
1711
+ // through the stable sort and the first-come overlap absorption below.
1712
+ // Measured on test/34's corpus, query "red": the two candidates (`red
1713
+ // circle` and `red square`) carry IDENTICAL `vote` and IDENTICAL `idfVote`
1714
+ // (1.3863 each, three seeds), so this comparator leaves them tied and the
1715
+ // stable sort keeps the ENUMERATION order — which is corpus-determined and
1716
+ // admits `red square`, 60/60 seeds. Adding an id tie-break (`|| a.anchor -
1717
+ // b.anchor`) picks `red circle` instead and makes a single region reach the
1718
+ // JOINT context, which is the premise `test/34` exists to protect. The
1719
+ // gates read IDF; this line only decides who gets looked at first.
1677
1720
  .sort((a, b) => b.vote - a.vote);
1678
1721
  const overlaps = (a: Attention, b: Attention) =>
1679
1722
  a.start < b.end && b.start < a.end;
@@ -1719,6 +1762,13 @@ export function commitVotes(
1719
1762
  rank,
1720
1763
  pooledVote: point.vote,
1721
1764
  idfVote: votesIdf.get(point.anchor) ?? 0,
1765
+ // The LARGEST single-region contribution behind this anchor — the bar
1766
+ // recall's own gate reads (mechanisms/recall.ts: forest[0].peak > LN2),
1767
+ // and until now the only decision-making quantity the climb computed and
1768
+ // did not publish. `regionPeak` reached `ranked` (see its build below)
1769
+ // and stopped there. Published, not recomputed: the value is the one the
1770
+ // climb already carries.
1771
+ peak: point.peak,
1722
1772
  candidateBreadth: regions.length,
1723
1773
  contributingVotes: regionAxioms.get(point.anchor) ?? 0,
1724
1774
  contributingEvidence: regionSupport.get(point.anchor) ?? 0,
@@ -1748,6 +1798,18 @@ export function commitVotes(
1748
1798
  let passesConsensusFloor: boolean | undefined;
1749
1799
  let pastLeadingSaturation: boolean | undefined;
1750
1800
  let tiedWithDominant: boolean | undefined;
1801
+ // ── ONE OF THREE ADMISSIONS, AND THEY ARE NOT THE SAME READING ────────
1802
+ // This block admits by VOTES: per-region evidence pooled, gated on the
1803
+ // natural break and on consensusFloor, with the dominant allowed to bypass
1804
+ // both. `structuralResonance` admits by a MARGIN over the estimator's own
1805
+ // noise, and `crossRegionVotes` admits by STRUCTURE (which regions may pair
1806
+ // at all, with at least one side individually discriminative). Read
1807
+ // together they look like one policy written three times; they are three
1808
+ // different measurements of the same question ("is this evidence?"), and
1809
+ // unifying them would average three readings into one — the mistake
1810
+ // `extraction.ts` records as "do not unify the two into one machine".
1811
+ // What they DO share, and must keep sharing, is the discipline of deriving
1812
+ // every bar from D/W/N rather than choosing it (thresholds.md).
1751
1813
  const rejectionReasons: AnchorRejectionReason[] = [];
1752
1814
  if (absorbed) {
1753
1815
  status = "overlap";
@@ -1757,9 +1819,75 @@ export function commitVotes(
1757
1819
  pastLeadingSaturation = pastLeading;
1758
1820
  const vote = votesIdf.get(point.anchor) ?? 0;
1759
1821
  if (roots.length === 0) {
1760
- // The first non-overlapping root is DOMINANT and bypasses the two
1761
- // vote thresholds (it always grounds) — only the leading-saturation
1762
- // gate still applies to it.
1822
+ // THE DOMINANCE PRIVILEGE, AND THE TENSION IT CARRIES (measured).
1823
+ //
1824
+ // The first non-overlapping candidate is DOMINANT: it bypasses both
1825
+ // vote gates below and grounds on its own; only the leading-saturation
1826
+ // gate still applies to it. The privilege is load-bearing — analogies,
1827
+ // substitutions and composed contexts are precisely candidates the
1828
+ // query does NOT contain, and the engine loses them without it.
1829
+ //
1830
+ // WHICH candidate receives it, though, is decided by this loop's ORDER.
1831
+ // That order comes from `ranked`, and when two candidates carry equal
1832
+ // evidence the stable sort preserves the ENUMERATION order, so the
1833
+ // privilege is allocated by an ordering rather than by a rule.
1834
+ //
1835
+ // Measured on test/34's corpus, query "red":
1836
+ //
1837
+ // 0:#77 vote=1.3863 idf=1.3863 [0,3) | 1:#49 vote=1.3863 idf=1.3863 [0,3)
1838
+ //
1839
+ // Both candidates (`red square` #77, `red circle` #49) have IDENTICAL
1840
+ // `vote` AND IDENTICAL `idfVote` over the SAME support span, so the
1841
+ // comparator leaves them tied, the second is absorbed as "overlap", and
1842
+ // the first grounds. With this build's enumeration order that first is
1843
+ // `red square`, 60/60 seeds, and `red circle` — the JOINT context — is
1844
+ // never reached by "red" alone. That is the premise test/34 exists to
1845
+ // protect: no single region reaches the joint context, which is what
1846
+ // makes the binding query unreachable without direct region
1847
+ // interaction.
1848
+ //
1849
+ // THE TENSION: the premise therefore holds BY ENUMERATION ORDER, not by
1850
+ // a rule, so any change to this ordering can move the privilege onto the
1851
+ // joint context and let one region reach it. Measured: adding
1852
+ // `|| a.anchor - b.anchor` — the lowest-id tie-break that AGENTS.md §2
1853
+ // sanctions as an equivalent corpus-determined tie-break — does exactly
1854
+ // that: "red" then attends to `red circle`, test/34 fails 6/1, and the
1855
+ // canonical suite reports 1 failure.
1856
+ //
1857
+ // TWO ATTEMPTS TO MAKE IT A RULE, BOTH REFUTED BY MEASUREMENT:
1858
+ //
1859
+ // 1. EVIDENCE SEPARATION. Grant the privilege only when the first
1860
+ // candidate's evidence is separated from the next distinct
1861
+ // candidate's by more than the co-dominant band (sqrt(k) *
1862
+ // estimatorNoise(D)). Refuted: that band exists to ADMIT the
1863
+ // anchors the estimator cannot separate from the dominant — its own
1864
+ // documented purpose — so withholding the privilege on ties removes
1865
+ // the very case it was written for. Suite: 4 failures (the two
1866
+ // co-dominant band laws, breadth/scale invariance, test/29 D2).
1867
+ //
1868
+ // 2. QUERY-OWNED CONTENT. Grant the privilege only to a candidate
1869
+ // that IS a recognised region's identity (regions.some(r => r.id ===
1870
+ // point.anchor)). Measured: for "circle" that identity IS the
1871
+ // ranked candidate, so the privilege stays and `circle` grounds; for
1872
+ // "red" the identity is the `red` node itself while the candidates
1873
+ // are the conjunctions, so neither is privileged; for "red then
1874
+ // circle" the composed context carries idf 3.958 and clears both
1875
+ // gates on its own evidence. All four control queries came out
1876
+ // right — and the suite: 10 failures, six of them in the
1877
+ // analogy/counterfactual/CAST suites ("an analogy still transfers
1878
+ // from a structure the query never names"; "a substitute the query
1879
+ // NAMES may still be voiced"). Refuted: the privilege exists to
1880
+ // admit what the query does NOT contain, so identity is the wrong
1881
+ // axis.
1882
+ //
1883
+ // WHAT A FUTURE ATTEMPT MUST RESPECT: whatever allocates this privilege
1884
+ // has to (a) keep it available to candidates the query does not contain
1885
+ // — analogies, substitutions, compositions — and (b) not depend on the
1886
+ // estimator's ordering among anchors of equal evidence, because that
1887
+ // ordering is not a fact about the corpus. No lever satisfying both has
1888
+ // been found. Until one is, this premise rests on the enumeration order
1889
+ // recorded above, and test/34 is the only test that notices if it
1890
+ // moves.
1763
1891
  dominant = true;
1764
1892
  if (pastLeading) {
1765
1893
  status = "root";
@@ -1768,8 +1896,15 @@ export function commitVotes(
1768
1896
  rejectionReasons.push("leading-saturation");
1769
1897
  }
1770
1898
  } else {
1771
- passesNaturalBreak = vote >= rootCut;
1772
- passesConsensusFloor = vote >= floor;
1899
+ // THE FLOOR AND THE BREAK READ THE IDF WEIGHTING. `floor` is derived
1900
+ // for pooled IDF-weighted votes, and `rootCut` comes from the IDF
1901
+ // distribution (`idfDesc`), so gating the mode-dependent `vote` against
1902
+ // either let a weighting mode change an admission (measured: anchor 87,
1903
+ // inverse 2.682 admitted vs direct 1.468 refused). Reading the IDF sum
1904
+ // makes the verdict mode-independent, and changes nothing in the
1905
+ // engine's own mode, where the two readings coincide.
1906
+ passesNaturalBreak = point.idfVote >= rootCut;
1907
+ passesConsensusFloor = point.idfVote >= floor;
1773
1908
  // CO-DOMINANT — an anchor the estimator cannot separate from the
1774
1909
  // dominant inherits the dominant's exemption, because that exemption's
1775
1910
  // only warrant is being TOP, and "top" is not a fact about the corpus
@@ -2467,6 +2602,13 @@ export async function structuralResonance(
2467
2602
  });
2468
2603
  };
2469
2604
 
2605
+ // ── ADMISSION BY MARGIN, not by votes (see voteRegions' note) ─────────
2606
+ // What this site measures: how far the best ANN proposal's effective score
2607
+ // (score × semanticConfidence) stands above the runner-up's, against
2608
+ // `estimatorNoise(D)`. What it does NOT measure: how many regions voted,
2609
+ // or whether the query's regions agree — that is voteRegions' question, and
2610
+ // here a synthetic gist has already replaced them. The two bars are both
2611
+ // derived (thresholds.md), and neither is a tuning of the other.
2470
2612
  let selected: StructuralResonanceProposal | null = null;
2471
2613
  let selectedReach: AncestorReach | null = null;
2472
2614
  let selectedIdf = 0;
@@ -2582,6 +2724,15 @@ async function crossRegionVotes(
2582
2724
  // successfully reconstructed while probing one pair must not be read and
2583
2725
  // perceived again while probing another pair in the same climb.
2584
2726
  const siblingGistMemo = new Map<number, CachedSiblingGist>();
2727
+ // ── ADMISSION BY STRUCTURE, not by a bar (see voteRegions' note) ──────
2728
+ // What this site decides: WHICH regions may pair at all — a region that
2729
+ // already voted (individually discriminative), or a KNOWN non-voting one as
2730
+ // the weak side of a pair whose other side voted; never two non-voting
2731
+ // regions, and never a span contained in a maximal one whose reading is
2732
+ // exact. The bar (the container's idf) comes later, on the candidate. So
2733
+ // its "rejection reasons" name structural disqualifications — a different
2734
+ // vocabulary because it answers a different question, and the three
2735
+ // taxonomies stay separate for the same reason the readings do.
2585
2736
  const votedSpans = new Set<string>();
2586
2737
  for (const rv of rvs.votes) votedSpans.add(`${rv.start},${rv.end}`);
2587
2738
  const seen = new Set<string>();
@@ -88,6 +88,49 @@ export function leafIdPrefix(
88
88
  return ids;
89
89
  }
90
90
 
91
+ /** Which prefixes of `prefix ‖ tail` are STORED NODES — as lengths in the
92
+ * tail's own coordinates, ascending, excluding the empty one. This is the
93
+ * candidate set a rule needs to join an already-stored prefix to a suffix it
94
+ * has not stored: a key names a relation exactly when `prefix ‖ tail[0..p]` IS
95
+ * a node, and a key can end strictly inside the tail without sitting on any
96
+ * fold boundary (a stored member's end is the end of ITS OWN stream, and the
97
+ * fold never emits a cut at a stream's end). Measured: "stockholm mayor"
98
+ * exists, leads on, and its boundary 6 is in neither the tail's cuts nor the
99
+ * concatenation's.
100
+ *
101
+ * ONE cheap content-addressed probe per offset — `leafIdPrefix` walks the bytes
102
+ * once (a point probe each), `findBranch` hashes the growing kid run — and NO
103
+ * `resolve`, which is what keeps this off the O(suffix) vector folds the
104
+ * recognition path pays. It stops at the first byte that was never interned,
105
+ * which costs nothing real: a stored key's bytes are interned by construction. */
106
+ export function keyEnds(
107
+ ctx: MindContext,
108
+ prefix: Uint8Array,
109
+ tail: Uint8Array,
110
+ ): number[] {
111
+ if (prefix.length === 0 || tail.length === 0) return [];
112
+ const joined = new Uint8Array(prefix.length + tail.length);
113
+ joined.set(prefix, 0);
114
+ joined.set(tail, prefix.length);
115
+ const ids = leafIdPrefix(ctx, joined);
116
+ if (ids.length < prefix.length) return [];
117
+ const ends: number[] = [];
118
+ // The kid run GROWS by one id per offset; `findBranch` wants an array, so the
119
+ // run is built once and pushed into, never re-sliced. Re-slicing
120
+ // `ids.slice(0, prefix.length + p)` per offset made this O(|tail| ·
121
+ // (|prefix| + |tail|)) — quadratic in the tail, where the learning path this
122
+ // follows slices a run that SHRINKS. Same ends, linear copying.
123
+ const run = ids.slice(0, prefix.length);
124
+ // The loop ENDS at the first byte that was never interned (`ids.length`):
125
+ // every later prefix contains it, so none of them can be a node either — this
126
+ // is where the scan stops, not a silent truncation of the answer.
127
+ for (let p = 1; prefix.length + p <= ids.length; p++) {
128
+ run.push(ids[prefix.length + p - 1]);
129
+ if (ctx.store.findBranch(run) !== null) ends.push(p);
130
+ }
131
+ return ends;
132
+ }
133
+
91
134
  /** The canonical W-window node ids of a byte stream, offset → id — the
92
135
  * CONTENT-ADDRESSED IDENTITY of every W-sized slice, under which any content
93
136
  * two deposits share IS the same node (hash-consing paid the comparison at
@@ -0,0 +1,202 @@
1
+ // corpus.ts — read the trained memory back out of the DAG, AS DATA.
2
+ //
3
+ // A trained experience pair IS one continuation edge: `src` is the context that
4
+ // was deposited, `dst` is what the mind learnt follows it. Reading them back
5
+ // uses the store's own structure and its own indexes — no auxiliary index is
6
+ // built, nothing is written, and NOTHING HERE KNOWS ABOUT TEXT: this layer takes
7
+ // bytes and returns bytes. The text case is one helper on the Mind
8
+ // (`searchCorpusText`), which encodes, calls this, and decodes.
9
+ //
10
+ // WHERE EACH STAGE COMES FROM (ported from the demo's `explore.ts`, which
11
+ // hand-rolled its own resolution):
12
+ //
13
+ // 1. PERCEIVE, CONTENT-ADDRESS and ADMIT the query — `recognise()`, the SAME
14
+ // machinery an answer goes through. It returns the sites: the query spans
15
+ // that content-addressed to a stored node that can lead somewhere. The
16
+ // demo's own recursive `findLeaf`/`findBranch` walk was a second
17
+ // implementation of exactly this, and it is NOT ported.
18
+ // 2. CLIMB the structural `kid` table from each resolved site to the
19
+ // edge-bearing contexts above it (`edgeAncestors`, traverse.ts), weighting
20
+ // each context by how much query content reached it.
21
+ // 3. READ the continuation off the edge table (`nextFirst`).
22
+ //
23
+ // COST is set by how much of the QUERY resolves, never by the size of the
24
+ // store: the sites are what recognition already found, the climb is bounded by
25
+ // the declared `corpusClimbs`/`corpusContextsPerClimb`, and every store call is
26
+ // a point probe or a capped read. All work is accounted by the store's own
27
+ // meter hooks when a response's meter is open — there is no second instrument.
28
+ //
29
+ // WHAT THIS IS NOT. Exact content addressing, not fuzzy keyword search: a query
30
+ // shares results with a stored note when it shares actual chunk-aligned content
31
+ // with it. An arbitrary mid-word fragment resolves to nothing, and the honest
32
+ // answer there is "nothing matched" — which is why `sampleCorpus` exists, and
33
+ // why the miss is reported as a STATE rather than as prose (the text helper
34
+ // turns it into words).
35
+
36
+ import { recognise } from "./recognition.js";
37
+ import { edgeAncestors } from "./traverse.js";
38
+ import type { MindContext } from "./types.js";
39
+
40
+ /** One stored experience pair, as bytes. */
41
+ export interface CorpusPair {
42
+ context: Uint8Array;
43
+ continuation: Uint8Array;
44
+ contextId: number;
45
+ continuationId: number;
46
+ /** Bytes of the query this pair was matched on — 0 when browsing. */
47
+ matchedBytes: number;
48
+ /** True when the stored bytes ran past the declared preview capacity. */
49
+ contextTruncated: boolean;
50
+ continuationTruncated: boolean;
51
+ }
52
+
53
+ /** Why a search produced no pairs. A STATE, so a caller's own layer can say it
54
+ * in its own words — the byte layer does not speak. */
55
+ export type CorpusMiss = "matched" | "nothing-resolved" | "no-continuations";
56
+
57
+ export interface CorpusResult {
58
+ pairs: CorpusPair[];
59
+ /** Query subtrees that content-addressed to a real stored node. */
60
+ resolved: number;
61
+ /** Distinct edge-bearing contexts the climb reached. */
62
+ reached: number;
63
+ /** Distinct contexts that carry a learnt continuation, store-wide. */
64
+ totalContexts: number;
65
+ /** True when these are browse samples rather than search results. */
66
+ browsed: boolean;
67
+ miss: CorpusMiss;
68
+ }
69
+
70
+ /** A context node as a pair, or null when it carries no continuation. */
71
+ function pairOf(
72
+ ctx: MindContext,
73
+ id: number,
74
+ matchedBytes: number,
75
+ ): CorpusPair | null {
76
+ const outs = ctx.store.nextFirst(id, 1);
77
+ if (outs.length === 0) return null;
78
+ const cap = ctx.cfg.corpusPreviewBytes;
79
+ const context = ctx.store.bytesPrefix(id, cap + 1);
80
+ const continuation = ctx.store.bytesPrefix(outs[0], cap + 1);
81
+ const contextTruncated = context.length > cap;
82
+ const continuationTruncated = continuation.length > cap;
83
+ return {
84
+ context: contextTruncated ? context.subarray(0, cap) : context,
85
+ continuation: continuationTruncated
86
+ ? continuation.subarray(0, cap)
87
+ : continuation,
88
+ contextId: id,
89
+ continuationId: outs[0],
90
+ matchedBytes,
91
+ contextTruncated,
92
+ continuationTruncated,
93
+ };
94
+ }
95
+
96
+ /** Which stored notes does this query reach? BYTES in, BYTES out.
97
+ *
98
+ * Exact content addressing through the machinery that already exists: the
99
+ * query's recognised sites are the resolved subtrees, the climb goes up from
100
+ * the biggest first, and a pair is a context that carries a continuation. */
101
+ export function searchCorpus(
102
+ ctx: MindContext,
103
+ queryBytes: Uint8Array,
104
+ limit?: number,
105
+ ): CorpusResult {
106
+ const store = ctx.store;
107
+ const want = Math.max(
108
+ 1,
109
+ Math.min(limit ?? ctx.cfg.corpusContextsPerClimb, ctx.cfg.corpusLimitMax),
110
+ );
111
+ // A resolved subtree must account for at least one window (W): a single
112
+ // character resolves against almost any store and means nothing. W is the
113
+ // mind's own line between chance and evidence — derived, never declared.
114
+ const floor = ctx.space.maxGroup;
115
+ const resolved = recognise(ctx, queryBytes).sites
116
+ .map((s) => ({ id: s.payload, len: store.contentLen(s.payload, 512) }))
117
+ .filter((r) => r.len >= floor);
118
+ // Biggest first, lowest id breaking ties: a clause is evidence, a character is
119
+ // noise, and equal evidence must not be decided by iteration order.
120
+ const byLength = resolved
121
+ .sort((a, b) => b.len - a.len || a.id - b.id)
122
+ .slice(0, ctx.cfg.corpusClimbs);
123
+
124
+ // Weight each context by how much query content reached it.
125
+ const weight = new Map<number, number>();
126
+ for (const { id, len } of byLength) {
127
+ for (
128
+ const root of edgeAncestors(ctx, id, ctx.cfg.corpusContextsPerClimb).roots
129
+ ) {
130
+ weight.set(root, (weight.get(root) ?? 0) + len);
131
+ }
132
+ }
133
+
134
+ const pairs: CorpusPair[] = [];
135
+ const ranked = [...weight.entries()].sort(
136
+ (a, b) => b[1] - a[1] || a[0] - b[0],
137
+ );
138
+ for (const [id, w] of ranked) {
139
+ if (pairs.length >= want) break;
140
+ const pair = pairOf(ctx, id, w);
141
+ if (pair) pairs.push(pair);
142
+ }
143
+ return {
144
+ pairs,
145
+ resolved: byLength.length,
146
+ reached: weight.size,
147
+ totalContexts: store.edgeSourceCount(),
148
+ browsed: false,
149
+ miss: pairs.length > 0
150
+ ? "matched"
151
+ : weight.size === 0
152
+ ? "nothing-resolved"
153
+ : "no-continuations",
154
+ };
155
+ }
156
+
157
+ /** Browse real pairs, striding the id space so the sample is spread rather than
158
+ * one local cluster. DETERMINISTIC: `from` is the caller's own offset, so
159
+ * browsing twice with different offsets shows different notes without a random
160
+ * draw (the demo drew `Math.random()`, which the engine cannot do — same seed,
161
+ * same order, same query must mean the same answer). */
162
+ export function sampleCorpus(
163
+ ctx: MindContext,
164
+ limit?: number,
165
+ from = 0,
166
+ ): CorpusResult {
167
+ const store = ctx.store;
168
+ const want = Math.max(
169
+ 1,
170
+ Math.min(limit ?? ctx.cfg.corpusContextsPerClimb, ctx.cfg.corpusLimitMax),
171
+ );
172
+ const total = store.nodeCount();
173
+ const probes = ctx.cfg.corpusSampleProbes;
174
+ const floorBytes = ctx.cfg.corpusSampleFloorBytes;
175
+ const pairs: CorpusPair[] = [];
176
+ // EACH CONTEXT AT MOST ONCE. Striding the id space revisits ids when the
177
+ // store is small relative to the probe budget (measured: a 160-node store
178
+ // returned the SAME pair six times for `limit: 6`), and a browse that repeats
179
+ // itself is not a browse. The demo had the same hole; it is invisible only on
180
+ // a store far larger than the probe budget.
181
+ const seen = new Set<number>();
182
+ for (let i = 0; i < probes && pairs.length < want && total > 0; i++) {
183
+ const slot = (i / probes + from) % 1;
184
+ const id = Math.floor(slot * total);
185
+ if (seen.has(id)) continue;
186
+ if (!store.has(id) || !store.hasNext(id)) continue;
187
+ if (store.contentLen(id, floorBytes) < floorBytes) continue;
188
+ const pair = pairOf(ctx, id, 0);
189
+ if (pair) {
190
+ seen.add(id);
191
+ pairs.push(pair);
192
+ }
193
+ }
194
+ return {
195
+ pairs,
196
+ resolved: 0,
197
+ reached: pairs.length,
198
+ totalContexts: store.edgeSourceCount(),
199
+ browsed: true,
200
+ miss: pairs.length > 0 ? "matched" : "no-continuations",
201
+ };
202
+ }