@hviana/sema 0.8.0 → 0.8.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (119) hide show
  1. package/AGENTS.md +22 -1
  2. package/DATASETS.md +1 -1
  3. package/dist/example/train_base/config.js +2 -2
  4. package/dist/example/train_base/corpora/massive.js +1 -1
  5. package/dist/example/train_base/readers.js +1 -1
  6. package/dist/src/config.d.ts +17 -0
  7. package/dist/src/config.js +18 -0
  8. package/dist/src/geometry.d.ts +10 -10
  9. package/dist/src/geometry.js +25 -24
  10. package/dist/src/meter.d.ts +29 -12
  11. package/dist/src/meter.js +58 -14
  12. package/dist/src/mind/attention.js +12 -12
  13. package/dist/src/mind/bridge.d.ts +8 -8
  14. package/dist/src/mind/bridge.js +33 -32
  15. package/dist/src/mind/corpus.d.ts +40 -0
  16. package/dist/src/mind/corpus.js +149 -0
  17. package/dist/src/mind/graph-search.d.ts +7 -8
  18. package/dist/src/mind/graph-search.js +244 -32
  19. package/dist/src/mind/index.d.ts +3 -1
  20. package/dist/src/mind/index.js +1 -0
  21. package/dist/src/mind/junction.d.ts +1 -1
  22. package/dist/src/mind/junction.js +8 -8
  23. package/dist/src/mind/learning.js +36 -35
  24. package/dist/src/mind/match.d.ts +8 -3
  25. package/dist/src/mind/match.js +156 -71
  26. package/dist/src/mind/mechanisms/cast.js +18 -2
  27. package/dist/src/mind/mechanisms/cover.js +19 -12
  28. package/dist/src/mind/mechanisms/prefix-completion.js +24 -24
  29. package/dist/src/mind/mechanisms/recall.js +38 -40
  30. package/dist/src/mind/mechanisms/reference.js +16 -16
  31. package/dist/src/mind/mind.d.ts +61 -7
  32. package/dist/src/mind/mind.js +72 -2
  33. package/dist/src/mind/pipeline-mechanism.d.ts +10 -8
  34. package/dist/src/mind/pipeline-mechanism.js +25 -21
  35. package/dist/src/mind/pipeline.d.ts +9 -9
  36. package/dist/src/mind/pipeline.js +49 -29
  37. package/dist/src/mind/primitives.d.ts +5 -5
  38. package/dist/src/mind/primitives.js +5 -5
  39. package/dist/src/mind/reasoning.d.ts +5 -1
  40. package/dist/src/mind/reasoning.js +54 -1
  41. package/dist/src/mind/recognition.d.ts +14 -13
  42. package/dist/src/mind/recognition.js +23 -23
  43. package/dist/src/mind/resonance.js +21 -21
  44. package/dist/src/mind/traverse.d.ts +54 -52
  45. package/dist/src/mind/traverse.js +83 -73
  46. package/dist/src/mind/types.d.ts +26 -4
  47. package/dist/src/store.d.ts +12 -12
  48. package/dist/src/store.js +12 -12
  49. package/docs/INDEX.md +2 -2
  50. package/docs/architecture/exact-vs-approximate.md +2 -1
  51. package/docs/architecture/fold-contract.md +1 -1
  52. package/docs/failures/tempting-but-wrong.md +33 -5
  53. package/docs/harness/gates.md +7 -7
  54. package/example/train_base/config.ts +2 -2
  55. package/example/train_base/corpora/massive.ts +1 -1
  56. package/example/train_base/readers.ts +1 -1
  57. package/jsr.json +1 -1
  58. package/package.json +1 -1
  59. package/src/config.ts +35 -0
  60. package/src/geometry.ts +25 -24
  61. package/src/meter.ts +61 -14
  62. package/src/mind/attention.ts +12 -12
  63. package/src/mind/bridge.ts +33 -32
  64. package/src/mind/corpus.ts +202 -0
  65. package/src/mind/graph-search.ts +261 -31
  66. package/src/mind/index.ts +8 -1
  67. package/src/mind/junction.ts +8 -8
  68. package/src/mind/learning.ts +36 -35
  69. package/src/mind/match.ts +163 -73
  70. package/src/mind/mechanisms/cast.ts +17 -1
  71. package/src/mind/mechanisms/cover.ts +18 -12
  72. package/src/mind/mechanisms/prefix-completion.ts +24 -24
  73. package/src/mind/mechanisms/recall.ts +38 -40
  74. package/src/mind/mechanisms/reference.ts +16 -16
  75. package/src/mind/mind.ts +129 -7
  76. package/src/mind/pipeline-mechanism.ts +25 -21
  77. package/src/mind/pipeline.ts +63 -38
  78. package/src/mind/primitives.ts +5 -5
  79. package/src/mind/reasoning.ts +55 -0
  80. package/src/mind/recognition.ts +23 -23
  81. package/src/mind/resonance.ts +21 -21
  82. package/src/mind/traverse.ts +83 -73
  83. package/src/mind/types.ts +30 -4
  84. package/src/store.ts +20 -20
  85. package/test/08-storage.test.mjs +1 -1
  86. package/test/100-complete-grounding-trace.test.mjs +109 -0
  87. package/test/101-alignment-gap-bound.test.mjs +106 -0
  88. package/test/102-production-composes-at-scale.test.mjs +110 -0
  89. package/test/103-alignment-gap-budget.test.mjs +89 -0
  90. package/test/104-composition-is-reported.test.mjs +90 -0
  91. package/test/105-derive-through-reports-its-refusal.test.mjs +113 -0
  92. package/test/106-the-join-fires.test.mjs +94 -0
  93. package/test/107-the-join-is-counted.test.mjs +81 -0
  94. package/test/108-the-join-chains.test.mjs +78 -0
  95. package/test/109-the-pivot-is-counted.test.mjs +60 -0
  96. package/test/110-the-reasoner-stops-when-the-question-is-answered.test.mjs +91 -0
  97. package/test/111-the-cover-assembly-is-counted.test.mjs +74 -0
  98. package/test/112-the-exploration-does-not-grow-with-the-hub.test.mjs +89 -0
  99. package/test/113-the-rationale-payload-is-bounded.test.mjs +84 -0
  100. package/test/114-alignment-budget-is-per-sweep.test.mjs +93 -0
  101. package/test/116-the-extension-is-gated-by-the-pipelines-own-remainder.test.mjs +100 -0
  102. package/test/117-corpus-search.test.mjs +171 -0
  103. package/test/14-scaling.test.mjs +10 -7
  104. package/test/35-prefix-edge.test.mjs +1 -1
  105. package/test/40-choosenext-scale-guard.test.mjs +16 -17
  106. package/test/56-bridge-identity-admission.test.mjs +6 -6
  107. package/test/70-prefix-completion.test.mjs +4 -3
  108. package/test/72-prefix-candidate-supply.test.mjs +3 -3
  109. package/test/73-scaffolding-only-bridge-abstains.test.mjs +6 -6
  110. package/test/75-multiturn-context-optimisation.test.mjs +5 -5
  111. package/test/76-reference-binding.test.mjs +6 -1
  112. package/test/84-composed-answer-honesty.test.mjs +5 -6
  113. package/test/88-dependency-footprint.test.mjs +1 -1
  114. package/test/89-completion-recursion.test.mjs +47 -19
  115. package/test/90-connector-read-cap.test.mjs +10 -8
  116. package/test/93-regime-prediction.test.mjs +10 -10
  117. package/test/94-cross-region-budget.test.mjs +2 -2
  118. package/test/95-wide-resonance-removed.test.mjs +8 -7
  119. package/test/96-bytes-walk-termination.test.mjs +3 -3
package/AGENTS.md CHANGED
@@ -134,7 +134,28 @@ silence). A simplification that fails an existing test is wrong until the test
134
134
  is proven wrong. Sublibraries test themselves in
135
135
  `src/{alu,derive,rabitq-ivf}/test/` with zero Sema dependency.
136
136
 
137
- ## 6. Dependencies and licensing
137
+ ## 6. Instrumentation — the meter and the rationale ARE the dev surface
138
+
139
+ `src/meter.ts` (what an answer COST) and `inspectRationale` (why it was CHOSEN,
140
+ `src/mind/rationale.ts`) are not debug helpers: they are Sema's development
141
+ instrumentation, and the only ones. Both are read through the public path —
142
+ `new Mind({ profile: true })` → `mind.lastCost` (`sumReports`/`formatReport`),
143
+ and the `inspectRationale` callback on `respond`/`respondText`/`respondTurn`.
144
+
145
+ When a change needs to be seen, measured, or proved, EXTEND THEM: a counter in
146
+ `meter.ts` (the one place a counter name exists — keep its four contracts true),
147
+ a step or note where the mechanism emits it (`src/mind/trace.ts` holds the move
148
+ vocabulary). A gap in instrumentation is a defect IN the instrumentation: close
149
+ it there, once, so the next person sees it too. Never add a parallel channel for
150
+ a single investigation — no ad-hoc logging or timing probes left in `src/`
151
+ (`performance.now()` belongs in `meter.ts`, not at a call site), no private
152
+ per-layer counter where a `meter.ts` field belongs, and no trace channel of your
153
+ own: a callback threaded through a call chain must FEED the rationale, the way
154
+ `GraphSearch`'s `onDerivation` feeds `traceDerivation`. (`store.ts`'s
155
+ `danglingReads`/`compactFailures` and the `console.warn`s that report them
156
+ predate this and stay: session-lifetime HEALTH counters, not per-response work.)
157
+
158
+ ## 7. Dependencies and licensing
138
159
 
139
160
  PolyForm Noncommercial 1.0.0 with separate commercial licensing (see
140
161
  `LICENSE.md`, `COMMERCIAL-LICENSE.md`, `TRADEMARKS.md`). The library has **no
package/DATASETS.md CHANGED
@@ -32,7 +32,7 @@ Two consequences follow, and both are load-bearing:
32
32
  2. **A corpus under a ShareAlike licence cannot enter the store**, because its
33
33
  copyleft would attach to the distributed artifact.
34
34
 
35
- Both rules are stated in [AGENTS.md](AGENTS.md) §6 and must be checked before
35
+ Both rules are stated in [AGENTS.md](AGENTS.md) §7 and must be checked before
36
36
  any corpus is added to a trainer.
37
37
 
38
38
  ---
@@ -4,8 +4,8 @@
4
4
  // the cache ceiling, the read budgets, the caps. A knob that describes ONE
5
5
  // CORPUS (which pairs of SmolSent, how many SODA dialogues, how long an Aya
6
6
  // field may be) belongs next to that corpus's adapter, together with the
7
- // evidence that fixed its default — see AGENTS.md §2.16: a comment carries the
8
- // constraint, and a constraint is only readable beside the code it constrains.
7
+ // evidence that fixed its default — a comment carries the constraint, and a
8
+ // constraint is only readable beside the code it constrains.
9
9
  import { join } from "node:path";
10
10
  /** Read an environment variable, or `d` when it is unset. */
11
11
  export const env = (k, d) => process.env[k] ?? d;
@@ -34,7 +34,7 @@ import { convertedParquetUnits } from "./converted-parquet.js";
34
34
  //
35
35
  // So it displaces some wrong answers and manufactures others, INCLUDING turning
36
36
  // a correct silence into a wrong answer — and honest silence is a stated
37
- // property of this engine (AGENTS §2.13). On the mixed-curriculum store the
37
+ // property of this engine (INVARIANTS.md). On the mixed-curriculum store the
38
38
  // same shape produced the fragment "nus" for "wake me up at nine am".
39
39
  //
40
40
  // That evidence is four probes on toy stores and is NOT conclusive; it is,
@@ -13,7 +13,7 @@
13
13
  //
14
14
  // THE ONLY THIRD-PARTY CODE IN THIS REPOSITORY IS BELOW, and it is LAZILY
15
15
  // LOADED. Sema itself imports nothing outside `node:` — that is a product
16
- // property, not an accident (AGENTS.md §6) — and this trainer is an EXAMPLE,
16
+ // property, not an accident (AGENTS.md §7) — and this trainer is an EXAMPLE,
17
17
  // not part of the library. hyparquet (+ its Snappy codec) is therefore a dev
18
18
  // dependency, and it is loaded by a dynamic import the first time a Parquet
19
19
  // corpus is actually read: a curriculum with no Parquet stage (SmolSent,
@@ -100,6 +100,23 @@ export interface MindConfig {
100
100
  seed: number;
101
101
  recallQueryK: number;
102
102
  haloQueryK: number;
103
+ /** Corpus reading (see src/mind/corpus.ts): results per call, resolved
104
+ * nodes climbed from, contexts requested per climb, probes used to stride
105
+ * the id space when browsing, bytes of each side a preview keeps, and the
106
+ * smallest deposited note browsing will show. Capacities and budgets only —
107
+ * the one material floor (a resolved node must account for W bytes) is
108
+ * derived from the geometry, not declared here. */
109
+ corpusLimitMax: number;
110
+ corpusClimbs: number;
111
+ corpusContextsPerClimb: number;
112
+ corpusSampleProbes: number;
113
+ corpusPreviewBytes: number;
114
+ corpusSampleFloorBytes: number;
115
+ /** Items one rationale step may ITEMISE (the whole field is still counted in
116
+ * the step's note). A capacity of the rationale, not of recall: sharing
117
+ * `recallQueryK` meant `new Mind({recallQueryK: 100000})` un-bounded the very
118
+ * payload the bound exists for (found by an adversarial review). */
119
+ rationaleSampleK: number;
103
120
  normalizeEpsilon: number;
104
121
  cosineEpsilon: number;
105
122
  alu: AluConfig;
@@ -5,6 +5,13 @@ export const DEFAULT_CONFIG = {
5
5
  seed: 42,
6
6
  recallQueryK: 12,
7
7
  haloQueryK: 12,
8
+ rationaleSampleK: 12,
9
+ corpusLimitMax: 24,
10
+ corpusClimbs: 24,
11
+ corpusContextsPerClimb: 6,
12
+ corpusSampleProbes: 6000,
13
+ corpusPreviewBytes: 220,
14
+ corpusSampleFloorBytes: 12,
8
15
  normalizeEpsilon: 1e-12,
9
16
  cosineEpsilon: 1e-12,
10
17
  alu: {
@@ -44,6 +51,17 @@ export function resolveConfig(opts = {}) {
44
51
  seed: opts.seed ?? DEFAULT_CONFIG.seed,
45
52
  recallQueryK: opts.recallQueryK ?? DEFAULT_CONFIG.recallQueryK,
46
53
  haloQueryK: opts.haloQueryK ?? DEFAULT_CONFIG.haloQueryK,
54
+ rationaleSampleK: opts.rationaleSampleK ?? DEFAULT_CONFIG.rationaleSampleK,
55
+ corpusLimitMax: opts.corpusLimitMax ?? DEFAULT_CONFIG.corpusLimitMax,
56
+ corpusClimbs: opts.corpusClimbs ?? DEFAULT_CONFIG.corpusClimbs,
57
+ corpusContextsPerClimb: opts.corpusContextsPerClimb ??
58
+ DEFAULT_CONFIG.corpusContextsPerClimb,
59
+ corpusSampleProbes: opts.corpusSampleProbes ??
60
+ DEFAULT_CONFIG.corpusSampleProbes,
61
+ corpusPreviewBytes: opts.corpusPreviewBytes ??
62
+ DEFAULT_CONFIG.corpusPreviewBytes,
63
+ corpusSampleFloorBytes: opts.corpusSampleFloorBytes ??
64
+ DEFAULT_CONFIG.corpusSampleFloorBytes,
47
65
  normalizeEpsilon: opts.normalizeEpsilon ?? DEFAULT_CONFIG.normalizeEpsilon,
48
66
  cosineEpsilon: opts.cosineEpsilon ?? DEFAULT_CONFIG.cosineEpsilon,
49
67
  alu: {
@@ -140,16 +140,16 @@ export declare function knownPrefixLength(bytes: Uint8Array, leafAt: (i: number)
140
140
  * correct boundary. Pass them through from `perceive`; the geometry
141
141
  * computes the stable prefix internally.
142
142
  *
143
- * `boundaries` is the CALLER-computed stable-prefix boundary set (§10.3):
144
- * strictly-increasing proper byte offsets, each the length of a prefix that
145
- * is already a stored whole-stream form. When given, the fold splits into
146
- * the segments between consecutive boundaries — each folded independently,
147
- * exactly as it folded when it was learned — and the segment roots join
148
- * LEFT-NESTED (((s₀·s₁)·s₂)…), so every learnt cumulative-context root
149
- * reappears as an identical subtree (and, by hash-consing, the very same
150
- * node) inside the grown stream. This is what lets a conversation's next
151
- * turn extend perception instead of refolding it: identical prefixes
152
- * produce identical subtrees regardless of what follows them. */
143
+ * `boundaries` is the CALLER-computed stable-prefix boundary set
144
+ * (fold-contract.md): strictly-increasing proper byte offsets, each the length
145
+ * of a prefix that is already a stored whole-stream form. When given, the fold
146
+ * splits into the segments between consecutive boundaries — each folded
147
+ * independently, exactly as it folded when it was learned — and the segment
148
+ * roots join LEFT-NESTED (((s₀·s₁)·s₂)…), so every learnt cumulative-context
149
+ * root reappears as an identical subtree (and, by hash-consing, the very same
150
+ * node) inside the grown stream. This is what lets a conversation's next turn
151
+ * extend perception instead of refolding it: identical prefixes produce
152
+ * identical subtrees regardless of what follows them. */
153
153
  export declare function bytesToTree(space: Space, alphabet: Alphabet, bytes: Uint8Array, leafAt?: (i: number) => number | null, lookup?: (leafIds: number[]) => number | null, boundaries?: readonly number[]): Sema;
154
154
  /** A plain content fold's reusable state: the level-0 cut edges over the whole
155
155
  * stream and each segment's independently-folded root. See
@@ -306,19 +306,20 @@ function bytesToLeaves(alphabet, bytes) {
306
306
  * sentences fall would be importing an assumption the architecture rejects.
307
307
  * Random binary must, and does, behave exactly like prose.
308
308
  *
309
- * Every constant is derived (§2.2): the cut mask is W, so a cut is offered once
310
- * per quantum of bytes — which, composed with the minimum below, puts the
311
- * expected segment at minLen + W − 1 ≈ 6 B rather than at W, deliberately (see
312
- * the refutation recorded at `cutRate` in {@link contentLevels}: a segment is
313
- * the flat PHRASE-scale unit the W-ary groups are built from, not a group of W
314
- * children, and forcing E[len] = W costs 15 tests). The minimum is W−1, `canonicalWindows`'s
315
- * straddle neighbour and the write side's own floor for a unit; and the maximum
316
- * is the KEYRING's seat count, because a segment folds as ONE flat node and
317
- * `fold` has exactly that many seats to bind children into. Capping there is
318
- * what keeps the fold light: a segment of 3..seats leaves is a single node,
319
- * where splitting it into W-groups plus a remainder would cost two or three
320
- * and the remainders barely share (measured: partial-arity nodes 504 → 3,590,
321
- * and total distinct nodes 8,142 → 9,712, when segments folded as [W][rest]). */
309
+ * Every constant is derived (thresholds.md): the cut mask is W, so a cut is
310
+ * offered once per quantum of bytes — which, composed with the minimum below,
311
+ * puts the expected segment at minLen + W − 1 ≈ 6 B rather than at W,
312
+ * deliberately (see the refutation recorded at `cutRate` in {@link
313
+ * contentLevels}: a segment is the flat PHRASE-scale unit the W-ary groups are
314
+ * built from, not a group of W children, and forcing E[len] = W costs 15
315
+ * tests). The minimum is W−1, `canonicalWindows`'s straddle neighbour and the
316
+ * write side's own floor for a unit; and the maximum is the KEYRING's seat
317
+ * count, because a segment folds as ONE flat node and `fold` has exactly that
318
+ * many seats to bind children into. Capping there is what keeps the fold light:
319
+ * a segment of 3..seats leaves is a single node, where splitting it into
320
+ * W-groups plus a remainder would cost two or three and the remainders barely
321
+ * share (measured: partial-arity nodes 504 → 3,590, and total distinct nodes
322
+ * 8,142 → 9,712, when segments folded as [W][rest]). */
322
323
  /** {@link contentBoundaries} plus, for each cut, its LEVEL — how deep in the
323
324
  * tree that cut reaches.
324
325
  *
@@ -568,16 +569,16 @@ export function knownPrefixLength(bytes, leafAt, lookup) {
568
569
  * correct boundary. Pass them through from `perceive`; the geometry
569
570
  * computes the stable prefix internally.
570
571
  *
571
- * `boundaries` is the CALLER-computed stable-prefix boundary set (§10.3):
572
- * strictly-increasing proper byte offsets, each the length of a prefix that
573
- * is already a stored whole-stream form. When given, the fold splits into
574
- * the segments between consecutive boundaries — each folded independently,
575
- * exactly as it folded when it was learned — and the segment roots join
576
- * LEFT-NESTED (((s₀·s₁)·s₂)…), so every learnt cumulative-context root
577
- * reappears as an identical subtree (and, by hash-consing, the very same
578
- * node) inside the grown stream. This is what lets a conversation's next
579
- * turn extend perception instead of refolding it: identical prefixes
580
- * produce identical subtrees regardless of what follows them. */
572
+ * `boundaries` is the CALLER-computed stable-prefix boundary set
573
+ * (fold-contract.md): strictly-increasing proper byte offsets, each the length
574
+ * of a prefix that is already a stored whole-stream form. When given, the fold
575
+ * splits into the segments between consecutive boundaries — each folded
576
+ * independently, exactly as it folded when it was learned — and the segment
577
+ * roots join LEFT-NESTED (((s₀·s₁)·s₂)…), so every learnt cumulative-context
578
+ * root reappears as an identical subtree (and, by hash-consing, the very same
579
+ * node) inside the grown stream. This is what lets a conversation's next turn
580
+ * extend perception instead of refolding it: identical prefixes produce
581
+ * identical subtrees regardless of what follows them. */
581
582
  export function bytesToTree(space, alphabet, bytes, leafAt, lookup, boundaries) {
582
583
  if (bytes.length === 0) {
583
584
  return sema(alphabet.vecs[0], new Uint8Array(0), null);
@@ -868,7 +869,7 @@ function flatFold(space, alphabet, bytes, from, to) {
868
869
  }
869
870
  return { tree: sema(gist, null, kids), len: n };
870
871
  }
871
- /** The stable-prefix segmented fold (§10.3). Each segment between
872
+ /* * The stable-prefix segmented fold (fold-contract.md). Each segment between
872
873
  * consecutive boundaries folds PLAINLY and independently; segment roots
873
874
  * join left-nested, and only the final root is normalized (the linear-fold
874
875
  * contract: one normalize per perception). A segment's own inner splits
@@ -48,9 +48,6 @@ export declare class Meter {
48
48
  nodeRecords: number;
49
49
  /** `store.bytes` / `store.bytesPrefix` — one reconstruction request. */
50
50
  byteReads: number;
51
- /** Bytes actually handed back by those reads — the real I/O volume, and
52
- * the number that exposes an unbounded read (AGENTS §2.8) that a call
53
- * count alone hides. */
54
51
  bytesRead: number;
55
52
  /** `store.contentLen`. */
56
53
  lenReads: number;
@@ -131,10 +128,10 @@ export declare class Meter {
131
128
  junctionPops: number;
132
129
  /** Ascents that ended by EXHAUSTING the expansion budget rather than by
133
130
  * deciding — the walk abstained and the caller silently fell through to a
134
- * lower tier of the ladder (§2.13: a degradation nothing else reports).
135
- * It rises the moment a SHARED budget is drained by an earlier walk, which
136
- * is what makes "this tier answered nothing" distinguishable from "this
137
- * tier never got to look". */
131
+ * lower tier of the ladder — honest degradation, and nothing else reports it
132
+ * (INVARIANTS.md). It rises the moment a SHARED budget is drained by an
133
+ * earlier walk, which is what makes "this tier answered nothing"
134
+ * distinguishable from "this tier never got to look". */
138
135
  junctionBudgetExhausted: number;
139
136
  /** Arbitrary byte spans whose distributional company was VSA-bundled from
140
137
  * existing episode halos. */
@@ -155,6 +152,31 @@ export declare class Meter {
155
152
  mechanismRuns: number;
156
153
  /** Candidates the decider weighed. */
157
154
  candidates: number;
155
+ /** `deriveThrough` yielded — a fact was reached through the subject the query
156
+ * never named. */
157
+ joinFired: number;
158
+ /** Refused: no key names the entity and the tail together. (A key that
159
+ * resolves but leads nowhere is not "refused" — it is not the relation, so
160
+ * the scan simply moves on; there is no counter for a case the loop cannot
161
+ * reach.) */
162
+ joinNoKey: number;
163
+ /** Refused: the fact contains no entity that leads anywhere. */
164
+ joinNoEntity: number;
165
+ /** Times the reasoner pivoted on a span its answer contains and stepped
166
+ * across that fact. */
167
+ pivotSteps: number;
168
+ /** `bridge` calls the cover makes assembling connectors (pairwise + n-ary). */
169
+ coverBridges: number;
170
+ /** Continuations a CHAIN hop offered the search. Bounded by the question
171
+ * (`ceil(queryLen / W)`) rather than by the corpus's fan-out — measured on a
172
+ * hub of degree 1083, offering every continuation grew the chart to 3113 outs
173
+ * and cost a 270 MB peak / 256 MB OOM for a two-word question. */
174
+ chainOffers: number;
175
+ /** Σ byte-allowance the n-ary interior passes those bridges. The allowance
176
+ * is `middleBytes + (m + 1) * W` — every intermediate answer's bytes plus
177
+ * one window of glue per joint — so it is the quantity that grows with a hub
178
+ * query's answers, and the first thing to read when the peak moves. */
179
+ coverAllowanceBytes: number;
158
180
  private readonly _phases;
159
181
  private readonly _t0;
160
182
  /** Every work counter's current value, by name — the snapshot `time`
@@ -163,11 +185,6 @@ export declare class Meter {
163
185
  /** Charge `ms`, one call, and a counter delta to a named phase.
164
186
  * Insertion-ordered, so a report reads in execution order. */
165
187
  charge(phase: string, ms: number, delta?: Record<string, number>): void;
166
- /** Time one SYNCHRONOUS phase. The sync/async seam (§2.10) is a real
167
- * contract — perception, recognition and the graph search are synchronous —
168
- * so a synchronous layer must not be wrapped in `time`'s promise just to be
169
- * measured: that would make the profiled path await where the unprofiled
170
- * one does not, and a meter never changes what a layer computes. */
171
188
  timeSync<T>(phase: string, fn: () => T): T;
172
189
  /** Time one async phase and attribute the work done inside it. Returns
173
190
  * the awaited value untouched — a meter never changes what a layer
package/dist/src/meter.js CHANGED
@@ -8,8 +8,8 @@
8
8
  // Four contracts, all load-bearing:
9
9
  //
10
10
  // 1. NEVER READ BY INFERENCE. No counter may reach a decision, a threshold,
11
- // or an ordering. Determinism (AGENTS §2.1) survives only because the
12
- // meter is write-only from the engine's point of view.
11
+ // or an ordering. Determinism (determinism.md) survives
12
+ // only because the meter is write-only from the engine's point of view.
13
13
  // 2. OFF BY DEFAULT, AND FREE WHEN OFF. Every call site is `meter?.x++` on
14
14
  // a null field. Nothing allocates, nothing is keyed, nothing is timed
15
15
  // unless a Meter is attached (`new Mind({ profile: true })`).
@@ -33,9 +33,9 @@ export class Meter {
33
33
  nodeRecords = 0;
34
34
  /** `store.bytes` / `store.bytesPrefix` — one reconstruction request. */
35
35
  byteReads = 0;
36
- /** Bytes actually handed back by those reads — the real I/O volume, and
37
- * the number that exposes an unbounded read (AGENTS §2.8) that a call
38
- * count alone hides. */
36
+ /* * Bytes actually handed back by those reads — the real I/O volume, and the
37
+ * number that exposes an unbounded read (bounded-reads.md) that a call count
38
+ * alone hides. */
39
39
  bytesRead = 0;
40
40
  /** `store.contentLen`. */
41
41
  lenReads = 0;
@@ -124,10 +124,10 @@ export class Meter {
124
124
  junctionPops = 0;
125
125
  /** Ascents that ended by EXHAUSTING the expansion budget rather than by
126
126
  * deciding — the walk abstained and the caller silently fell through to a
127
- * lower tier of the ladder (§2.13: a degradation nothing else reports).
128
- * It rises the moment a SHARED budget is drained by an earlier walk, which
129
- * is what makes "this tier answered nothing" distinguishable from "this
130
- * tier never got to look". */
127
+ * lower tier of the ladder — honest degradation, and nothing else reports it
128
+ * (INVARIANTS.md). It rises the moment a SHARED budget is drained by an
129
+ * earlier walk, which is what makes "this tier answered nothing"
130
+ * distinguishable from "this tier never got to look". */
131
131
  junctionBudgetExhausted = 0;
132
132
  /** Arbitrary byte spans whose distributional company was VSA-bundled from
133
133
  * existing episode halos. */
@@ -149,6 +149,50 @@ export class Meter {
149
149
  mechanismRuns = 0;
150
150
  /** Candidates the decider weighed. */
151
151
  candidates = 0;
152
+ // ── Graph search: the fact join (DIRECTION) ─────────────────────────────
153
+ //
154
+ // The join's outcome was observable ONLY through the rationale, and the
155
+ // rationale PERTURBS the search (measured: appending text to a refusal note
156
+ // changed a traced answer). These four counters are the untraced view — the
157
+ // same surface every other work counter uses, incremented where the decision
158
+ // is made, never behind a trace guard.
159
+ /** `deriveThrough` yielded — a fact was reached through the subject the query
160
+ * never named. */
161
+ joinFired = 0;
162
+ /** Refused: no key names the entity and the tail together. (A key that
163
+ * resolves but leads nowhere is not "refused" — it is not the relation, so
164
+ * the scan simply moves on; there is no counter for a case the loop cannot
165
+ * reach.) */
166
+ joinNoKey = 0;
167
+ /** Refused: the fact contains no entity that leads anywhere. */
168
+ joinNoEntity = 0;
169
+ // ── Mind: the multi-hop pivot (EXTENSION) ───────────────────────────────
170
+ //
171
+ // `pivotStep` was observable only through the rationale, and the rationale
172
+ // perturbs the search (measured). How far the reasoner hopped is a
173
+ // BEHAVIOUR, so it needs an untraced view: one counter, incremented where the
174
+ // step is emitted.
175
+ /** Times the reasoner pivoted on a span its answer contains and stepped
176
+ * across that fact. */
177
+ pivotSteps = 0;
178
+ // ── Mind: the cover's connector assembly (LIMIT) ────────────────────────
179
+ //
180
+ // The cover's `run` is 91% of a hub query's time (`"Hello."`: 2.7 s of 3.0 s)
181
+ // and holds its ~270 MB peak, and none of it was countable: `searchPushes`
182
+ // and `candidates` do not see the connector assembly. These two counters are
183
+ // the untraced view of it.
184
+ /** `bridge` calls the cover makes assembling connectors (pairwise + n-ary). */
185
+ coverBridges = 0;
186
+ /** Continuations a CHAIN hop offered the search. Bounded by the question
187
+ * (`ceil(queryLen / W)`) rather than by the corpus's fan-out — measured on a
188
+ * hub of degree 1083, offering every continuation grew the chart to 3113 outs
189
+ * and cost a 270 MB peak / 256 MB OOM for a two-word question. */
190
+ chainOffers = 0;
191
+ /** Σ byte-allowance the n-ary interior passes those bridges. The allowance
192
+ * is `middleBytes + (m + 1) * W` — every intermediate answer's bytes plus
193
+ * one window of glue per joint — so it is the quantity that grows with a hub
194
+ * query's answers, and the first thing to read when the peak moves. */
195
+ coverAllowanceBytes = 0;
152
196
  // ── Phases ──────────────────────────────────────────────────────────────
153
197
  _phases = new Map();
154
198
  _t0 = performance.now();
@@ -180,11 +224,11 @@ export class Meter {
180
224
  }
181
225
  }
182
226
  }
183
- /** Time one SYNCHRONOUS phase. The sync/async seam (§2.10) is a real
184
- * contract — perception, recognition and the graph search are synchronous —
185
- * so a synchronous layer must not be wrapped in `time`'s promise just to be
186
- * measured: that would make the profiled path await where the unprofiled
187
- * one does not, and a meter never changes what a layer computes. */
227
+ /* * Time one SYNCHRONOUS phase. The sync/async seam is a real contract
228
+ * (meter.md) — perception, recognition and the graph search are synchronous —
229
+ * so a synchronous layer must not be wrapped in `time`'s promise just to be
230
+ * measured: that would make the profiled path await where the unprofiled one
231
+ * does not, and a meter never changes what a layer computes. */
188
232
  timeSync(phase, fn) {
189
233
  const before = this.snapshot();
190
234
  const t = performance.now();
@@ -1364,10 +1364,10 @@ export function canonicalChunkId(ctx, regionBytes, N, reachMemo) {
1364
1364
  // CAST lost a point of attention it needed (test/29 D1/D2).
1365
1365
  //
1366
1366
  // So scan every offset and prefer an anchor that still discriminates: not
1367
- // saturated, and among those the one reaching the FEWEST contexts (§2.7,
1368
- // corpus-global). Only when every window in the region saturates does the
1369
- // old generalising choice stand — there is then no discriminative anchor to
1370
- // find, and abstaining is the honest outcome.
1367
+ // saturated, and among those the one reaching the FEWEST contexts
1368
+ // (commonality.md, corpus-global). Only when every window in the region
1369
+ // saturates does the old generalising choice stand — there is then no
1370
+ // discriminative anchor to find, and abstaining is the honest outcome.
1371
1371
  let discId = null;
1372
1372
  let discReached = Infinity;
1373
1373
  let fallback = null;
@@ -1849,16 +1849,16 @@ async function crossRegionVotes(ctx, query, regions, rvs, k, N, reachMemo, td) {
1849
1849
  const consumed = new Set();
1850
1850
  let probes = 0;
1851
1851
  // When atoms themselves are hubs (atomIsHub — a single byte reaches ≥ √N
1852
- // contexts, §2.8's own predicate), the corpus is large enough that the
1853
- // cross-region junction walks are dominated by the drift through common
1854
- // content's ancestry. Each of k candidate pairs otherwise spends its own
1855
- // √N·W budget (profiled: 160,210 junction pops, 31% of think at
1856
- // N = 325,608), and a cumulative dialogue multiplies bounded work into tens
1857
- // of seconds. The structural walk is therefore given ONE k·W allowance per
1852
+ // contexts, bounded-reads.md's own predicate), the corpus is large enough
1853
+ // that the cross-region junction walks are dominated by the drift through
1854
+ // common content's ancestry. Each of k candidate pairs otherwise spends its
1855
+ // own √N·W budget (profiled: 160,210 junction pops, 31% of think at N =
1856
+ // 325,608), and a cumulative dialogue multiplies bounded work into tens of
1857
+ // seconds. The structural walk is therefore given ONE k·W allowance per
1858
1858
  // evidence tier, shared across every pair — k pairs × W phrase-scale levels,
1859
1859
  // the minimal exact check; a pair whose container is not reached within it
1860
- // falls through to the resonance tier (the ANN proposes what the shallow
1861
- // walk no longer exhaustively scans, §2.3).
1860
+ // falls through to the resonance tier (the ANN proposes what the shallow walk
1861
+ // no longer exhaustively scans, exact-vs-approximate.md).
1862
1862
  //
1863
1863
  // Below atomIsHub the store is small and atoms still discriminate, so the
1864
1864
  // walks keep exhaustive exact traversal (per-walk √N·W) — the shared budget
@@ -25,13 +25,13 @@ export declare function dismissedKnownContent(ctx: MindContext, query: Uint8Arra
25
25
  /** Recall's corroborated-substitution bridge — see the module comment.
26
26
  * Returns the best bridged grounding proposal, or null. */
27
27
  /** `proposed` is a THUNK, not a list: the bridge's own cheap gates (the
28
- * two-quantum query floor and the O(|query|) stored-window anchor scan)
29
- * decide whether ANY candidate can be aligned, and they need no proposals
30
- * to do it. Resolving the caller's proposals eagerly meant recall paid its
31
- * exhaustive whole-index resonance — the most expensive single act on the
32
- * refusal path — for every query, including the ones whose windows the
33
- * store has never seen and which the anchor scan rejects outright. Same
34
- * investment discipline the mechanism floors follow (AGENTS §2.6): never
35
- * compute a shared analysis just to discard it. */
28
+ * two-quantum query floor and the O(|query|) stored-window anchor scan) decide
29
+ * whether ANY candidate can be aligned, and they need no proposals to do it.
30
+ * Resolving the caller's proposals eagerly meant recall paid its exhaustive
31
+ * whole-index resonance — the most expensive single act on the refusal path —
32
+ * for every query, including the ones whose windows the store has never seen
33
+ * and which the anchor scan rejects outright. Same investment discipline the
34
+ * mechanism floors follow (mechanism-market.md): never compute a shared
35
+ * analysis just to discard it. */
36
36
  export declare function substitutionBridge(ctx: MindContext, query: Uint8Array, proposed?: () => Promise<ReadonlyArray<number>>): Promise<BridgeHit | null>;
37
37
  export {};
@@ -121,22 +121,22 @@ export function dismissedKnownContent(ctx, query, spans) {
121
121
  }
122
122
  return false;
123
123
  }
124
- // The seeded aligner this file used to own now lives in the shared match
125
- // family as {@link alignAround} — the frame reading (match.ts) reads the same
126
- // gaps and asks the OPPOSITE question of them (see AlignGap's own doc). Two
127
- // consumers, one definition (AGENTS §2.5); the bridge's reading is unchanged.
124
+ // The seeded aligner this file used to own now lives in the shared match family
125
+ // as {@link alignAround} — the frame reading (match.ts) reads the same gaps and
126
+ // asks the OPPOSITE question of them (see AlignGap's own doc). Two consumers,
127
+ // one definition (factored-machinery.md); the bridge's reading is unchanged.
128
128
  const align = alignAround;
129
129
  /** Recall's corroborated-substitution bridge — see the module comment.
130
130
  * Returns the best bridged grounding proposal, or null. */
131
131
  /** `proposed` is a THUNK, not a list: the bridge's own cheap gates (the
132
- * two-quantum query floor and the O(|query|) stored-window anchor scan)
133
- * decide whether ANY candidate can be aligned, and they need no proposals
134
- * to do it. Resolving the caller's proposals eagerly meant recall paid its
135
- * exhaustive whole-index resonance — the most expensive single act on the
136
- * refusal path — for every query, including the ones whose windows the
137
- * store has never seen and which the anchor scan rejects outright. Same
138
- * investment discipline the mechanism floors follow (AGENTS §2.6): never
139
- * compute a shared analysis just to discard it. */
132
+ * two-quantum query floor and the O(|query|) stored-window anchor scan) decide
133
+ * whether ANY candidate can be aligned, and they need no proposals to do it.
134
+ * Resolving the caller's proposals eagerly meant recall paid its exhaustive
135
+ * whole-index resonance — the most expensive single act on the refusal path —
136
+ * for every query, including the ones whose windows the store has never seen
137
+ * and which the anchor scan rejects outright. Same investment discipline the
138
+ * mechanism floors follow (mechanism-market.md): never compute a shared
139
+ * analysis just to discard it. */
140
140
  export async function substitutionBridge(ctx, query, proposed = async () => []) {
141
141
  const meter = ctx.meter;
142
142
  return meter
@@ -238,12 +238,12 @@ async function bridgeImpl(ctx, query, proposed) {
238
238
  ctx.trace?.step("substitutionBridge", [rItem(query, "query")], [], "no stored query window can anchor a corroborated substitution", undefined, diagnostics);
239
239
  return null;
240
240
  }
241
- // NO DISCRIMINATING LITERAL EVIDENCE — abstain (§2.13). A bridge grounds
242
- // through the literal spans it did NOT substitute; those anchors are the
243
- // whole of its evidence. When every one of them is SATURATED — containment
244
- // clamped at the √N hub bound, i.e. the window is corpus-global scaffolding
245
- // — the query's unsubstituted part discriminates nothing, and the single
246
- // substituted span is carrying the entire semantic load. That is not a
241
+ // NO DISCRIMINATING LITERAL EVIDENCE — abstain (INVARIANTS.md). A bridge
242
+ // grounds through the literal spans it did NOT substitute; those anchors are
243
+ // the whole of its evidence. When every one of them is SATURATED —
244
+ // containment clamped at the √N hub bound, i.e. the window is corpus-global
245
+ // scaffolding — the query's unsubstituted part discriminates nothing, and the
246
+ // single substituted span is carrying the entire semantic load. That is not a
247
247
  // corroborated bridge; it is a template match, and it FABRICATES.
248
248
  //
249
249
  // Measured on the trained store (hubBound 571). "What is the capital of"
@@ -258,7 +258,8 @@ async function bridgeImpl(ctx, query, proposed) {
258
258
  // them silent and cannot be credited for them.
259
259
  //
260
260
  // This introduces NO new threshold: `bound` is the same √N reading of "hub"
261
- // the anchor scan already clamps its own containment read to (§2.2, §2.7).
261
+ // the anchor scan already clamps its own containment read to (thresholds.md,
262
+ // commonality.md).
262
263
  if (allWindowsAreScaffolding(ctx, query)) {
263
264
  ctx.trace?.step("substitutionBridge", [rItem(query, "query")], [], "every query window that could anchor is corpus-global scaffolding — " +
264
265
  "no literal evidence to corroborate a substitution", undefined, diagnostics);
@@ -297,8 +298,8 @@ async function bridgeImpl(ctx, query, proposed) {
297
298
  //
298
299
  // The question every gap poses is "may the two forms differ HERE without
299
300
  // differing in what they SAY?", and that is the discriminative-vs-
300
- // scaffolding question AGENTS §2.7 names, over the CORPUS-GLOBAL
301
- // population. It already has one definition — `dominates(reachOf(...), N)`,
301
+ // scaffolding question commonality.md names, over the CORPUS-GLOBAL
302
+ // population. It already has one definition — `dominates(reachOf(...), N)`,
302
303
  // the same gate confluence's filler test uses ("scaffolding never binds").
303
304
  // Nothing new is derived here; the bar is read, not invented.
304
305
  //
@@ -310,17 +311,17 @@ async function bridgeImpl(ctx, query, proposed) {
310
311
  // climb's own definition of non-discriminative), or it resolves to a
311
312
  // majority of the corpus's contexts. "the process of ", " is the ".
312
313
  //
313
- // THE READING MATTERS, not just the population (AGENTS §2.7). This
314
- // deliberately does NOT go through `reachOf`, which maps BOTH "saturated"
315
- // and "reaches nothing" to Infinity. For IDF weighting those are the same
316
- // thing (no usable identity evidence); for THIS question they are
317
- // opposites — a window reaching nothing is novel content, the most
318
- // discriminative material there is, and reading it as Infinity would call
319
- // it scaffolding. Measured: with `reachOf`, "Is water wet?" was answered
320
- // with "No, heavy water is not wet." — "heav"/"eavy" occur once, reach no
321
- // edge-bearing ancestor, and were written off as filler. So an
322
- // empty-rooted window is NEVER explained, and neither is an untrained one
323
- // (the same principle attestedQ applies to the query side).
314
+ // THE READING MATTERS, not just the population — see commonality.md. This
315
+ // deliberately does NOT go through `reachOf`, which maps BOTH "saturated" and
316
+ // "reaches nothing" to Infinity. For IDF weighting those are the same thing
317
+ // (no usable identity evidence); for THIS question they are opposites — a
318
+ // window reaching nothing is novel content, the most discriminative material
319
+ // there is, and reading it as Infinity would call it scaffolding. Measured:
320
+ // with `reachOf`, "Is water wet?" was answered with "No, heavy water is not
321
+ // wet." — "heav"/"eavy" occur once, reach no edge-bearing ancestor, and were
322
+ // written off as filler. So an empty-rooted window is NEVER explained, and
323
+ // neither is an untrained one (the same principle attestedQ applies to the
324
+ // query side).
324
325
  const reachMemo = sharedReachMemo(ctx);
325
326
  const explainedSpan = (bytes, from, to) => {
326
327
  if (to - from < W)
@@ -0,0 +1,40 @@
1
+ import type { MindContext } from "./types.js";
2
+ /** One stored experience pair, as bytes. */
3
+ export interface CorpusPair {
4
+ context: Uint8Array;
5
+ continuation: Uint8Array;
6
+ contextId: number;
7
+ continuationId: number;
8
+ /** Bytes of the query this pair was matched on — 0 when browsing. */
9
+ matchedBytes: number;
10
+ /** True when the stored bytes ran past the declared preview capacity. */
11
+ contextTruncated: boolean;
12
+ continuationTruncated: boolean;
13
+ }
14
+ /** Why a search produced no pairs. A STATE, so a caller's own layer can say it
15
+ * in its own words — the byte layer does not speak. */
16
+ export type CorpusMiss = "matched" | "nothing-resolved" | "no-continuations";
17
+ export interface CorpusResult {
18
+ pairs: CorpusPair[];
19
+ /** Query subtrees that content-addressed to a real stored node. */
20
+ resolved: number;
21
+ /** Distinct edge-bearing contexts the climb reached. */
22
+ reached: number;
23
+ /** Distinct contexts that carry a learnt continuation, store-wide. */
24
+ totalContexts: number;
25
+ /** True when these are browse samples rather than search results. */
26
+ browsed: boolean;
27
+ miss: CorpusMiss;
28
+ }
29
+ /** Which stored notes does this query reach? BYTES in, BYTES out.
30
+ *
31
+ * Exact content addressing through the machinery that already exists: the
32
+ * query's recognised sites are the resolved subtrees, the climb goes up from
33
+ * the biggest first, and a pair is a context that carries a continuation. */
34
+ export declare function searchCorpus(ctx: MindContext, queryBytes: Uint8Array, limit?: number): CorpusResult;
35
+ /** Browse real pairs, striding the id space so the sample is spread rather than
36
+ * one local cluster. DETERMINISTIC: `from` is the caller's own offset, so
37
+ * browsing twice with different offsets shows different notes without a random
38
+ * draw (the demo drew `Math.random()`, which the engine cannot do — same seed,
39
+ * same order, same query must mean the same answer). */
40
+ export declare function sampleCorpus(ctx: MindContext, limit?: number, from?: number): CorpusResult;