@hviana/sema 0.8.1 → 0.8.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (57) hide show
  1. package/dist/src/config.d.ts +17 -0
  2. package/dist/src/config.js +18 -0
  3. package/dist/src/meter.d.ts +25 -0
  4. package/dist/src/meter.js +44 -0
  5. package/dist/src/mind/corpus.d.ts +40 -0
  6. package/dist/src/mind/corpus.js +149 -0
  7. package/dist/src/mind/graph-search.d.ts +7 -0
  8. package/dist/src/mind/graph-search.js +235 -24
  9. package/dist/src/mind/index.d.ts +3 -1
  10. package/dist/src/mind/index.js +1 -0
  11. package/dist/src/mind/match.d.ts +8 -3
  12. package/dist/src/mind/match.js +142 -58
  13. package/dist/src/mind/mechanisms/cast.js +18 -2
  14. package/dist/src/mind/mechanisms/cover.js +6 -0
  15. package/dist/src/mind/mind.d.ts +55 -0
  16. package/dist/src/mind/mind.js +72 -2
  17. package/dist/src/mind/pipeline.js +25 -6
  18. package/dist/src/mind/reasoning.d.ts +5 -1
  19. package/dist/src/mind/reasoning.js +54 -1
  20. package/dist/src/mind/traverse.js +9 -1
  21. package/dist/src/mind/types.d.ts +22 -0
  22. package/docs/failures/tempting-but-wrong.md +31 -2
  23. package/jsr.json +1 -1
  24. package/package.json +1 -1
  25. package/src/config.ts +35 -0
  26. package/src/meter.ts +47 -0
  27. package/src/mind/corpus.ts +202 -0
  28. package/src/mind/graph-search.ts +252 -23
  29. package/src/mind/index.ts +8 -1
  30. package/src/mind/match.ts +143 -54
  31. package/src/mind/mechanisms/cast.ts +17 -1
  32. package/src/mind/mechanisms/cover.ts +5 -0
  33. package/src/mind/mind.ts +123 -0
  34. package/src/mind/pipeline.ts +30 -6
  35. package/src/mind/reasoning.ts +55 -0
  36. package/src/mind/traverse.ts +9 -1
  37. package/src/mind/types.ts +26 -0
  38. package/test/100-complete-grounding-trace.test.mjs +109 -0
  39. package/test/101-alignment-gap-bound.test.mjs +106 -0
  40. package/test/102-production-composes-at-scale.test.mjs +110 -0
  41. package/test/103-alignment-gap-budget.test.mjs +89 -0
  42. package/test/104-composition-is-reported.test.mjs +90 -0
  43. package/test/105-derive-through-reports-its-refusal.test.mjs +113 -0
  44. package/test/106-the-join-fires.test.mjs +94 -0
  45. package/test/107-the-join-is-counted.test.mjs +81 -0
  46. package/test/108-the-join-chains.test.mjs +78 -0
  47. package/test/109-the-pivot-is-counted.test.mjs +60 -0
  48. package/test/110-the-reasoner-stops-when-the-question-is-answered.test.mjs +91 -0
  49. package/test/111-the-cover-assembly-is-counted.test.mjs +74 -0
  50. package/test/112-the-exploration-does-not-grow-with-the-hub.test.mjs +89 -0
  51. package/test/113-the-rationale-payload-is-bounded.test.mjs +84 -0
  52. package/test/114-alignment-budget-is-per-sweep.test.mjs +93 -0
  53. package/test/116-the-extension-is-gated-by-the-pipelines-own-remainder.test.mjs +100 -0
  54. package/test/117-corpus-search.test.mjs +171 -0
  55. package/test/14-scaling.test.mjs +10 -7
  56. package/test/76-reference-binding.test.mjs +6 -1
  57. package/test/89-completion-recursion.test.mjs +30 -5
@@ -100,6 +100,23 @@ export interface MindConfig {
100
100
  seed: number;
101
101
  recallQueryK: number;
102
102
  haloQueryK: number;
103
+ /** Corpus reading (see src/mind/corpus.ts): results per call, resolved
104
+ * nodes climbed from, contexts requested per climb, probes used to stride
105
+ * the id space when browsing, bytes of each side a preview keeps, and the
106
+ * smallest deposited note browsing will show. Capacities and budgets only —
107
+ * the one material floor (a resolved node must account for W bytes) is
108
+ * derived from the geometry, not declared here. */
109
+ corpusLimitMax: number;
110
+ corpusClimbs: number;
111
+ corpusContextsPerClimb: number;
112
+ corpusSampleProbes: number;
113
+ corpusPreviewBytes: number;
114
+ corpusSampleFloorBytes: number;
115
+ /** Items one rationale step may ITEMISE (the whole field is still counted in
116
+ * the step's note). A capacity of the rationale, not of recall: sharing
117
+ * `recallQueryK` meant `new Mind({recallQueryK: 100000})` un-bounded the very
118
+ * payload the bound exists for (found by an adversarial review). */
119
+ rationaleSampleK: number;
103
120
  normalizeEpsilon: number;
104
121
  cosineEpsilon: number;
105
122
  alu: AluConfig;
@@ -5,6 +5,13 @@ export const DEFAULT_CONFIG = {
5
5
  seed: 42,
6
6
  recallQueryK: 12,
7
7
  haloQueryK: 12,
8
+ rationaleSampleK: 12,
9
+ corpusLimitMax: 24,
10
+ corpusClimbs: 24,
11
+ corpusContextsPerClimb: 6,
12
+ corpusSampleProbes: 6000,
13
+ corpusPreviewBytes: 220,
14
+ corpusSampleFloorBytes: 12,
8
15
  normalizeEpsilon: 1e-12,
9
16
  cosineEpsilon: 1e-12,
10
17
  alu: {
@@ -44,6 +51,17 @@ export function resolveConfig(opts = {}) {
44
51
  seed: opts.seed ?? DEFAULT_CONFIG.seed,
45
52
  recallQueryK: opts.recallQueryK ?? DEFAULT_CONFIG.recallQueryK,
46
53
  haloQueryK: opts.haloQueryK ?? DEFAULT_CONFIG.haloQueryK,
54
+ rationaleSampleK: opts.rationaleSampleK ?? DEFAULT_CONFIG.rationaleSampleK,
55
+ corpusLimitMax: opts.corpusLimitMax ?? DEFAULT_CONFIG.corpusLimitMax,
56
+ corpusClimbs: opts.corpusClimbs ?? DEFAULT_CONFIG.corpusClimbs,
57
+ corpusContextsPerClimb: opts.corpusContextsPerClimb ??
58
+ DEFAULT_CONFIG.corpusContextsPerClimb,
59
+ corpusSampleProbes: opts.corpusSampleProbes ??
60
+ DEFAULT_CONFIG.corpusSampleProbes,
61
+ corpusPreviewBytes: opts.corpusPreviewBytes ??
62
+ DEFAULT_CONFIG.corpusPreviewBytes,
63
+ corpusSampleFloorBytes: opts.corpusSampleFloorBytes ??
64
+ DEFAULT_CONFIG.corpusSampleFloorBytes,
47
65
  normalizeEpsilon: opts.normalizeEpsilon ?? DEFAULT_CONFIG.normalizeEpsilon,
48
66
  cosineEpsilon: opts.cosineEpsilon ?? DEFAULT_CONFIG.cosineEpsilon,
49
67
  alu: {
@@ -152,6 +152,31 @@ export declare class Meter {
152
152
  mechanismRuns: number;
153
153
  /** Candidates the decider weighed. */
154
154
  candidates: number;
155
+ /** `deriveThrough` yielded — a fact was reached through the subject the query
156
+ * never named. */
157
+ joinFired: number;
158
+ /** Refused: no key names the entity and the tail together. (A key that
159
+ * resolves but leads nowhere is not "refused" — it is not the relation, so
160
+ * the scan simply moves on; there is no counter for a case the loop cannot
161
+ * reach.) */
162
+ joinNoKey: number;
163
+ /** Refused: the fact contains no entity that leads anywhere. */
164
+ joinNoEntity: number;
165
+ /** Times the reasoner pivoted on a span its answer contains and stepped
166
+ * across that fact. */
167
+ pivotSteps: number;
168
+ /** `bridge` calls the cover makes assembling connectors (pairwise + n-ary). */
169
+ coverBridges: number;
170
+ /** Continuations a CHAIN hop offered the search. Bounded by the question
171
+ * (`ceil(queryLen / W)`) rather than by the corpus's fan-out — measured on a
172
+ * hub of degree 1083, offering every continuation grew the chart to 3113 outs
173
+ * and cost a 270 MB peak / 256 MB OOM for a two-word question. */
174
+ chainOffers: number;
175
+ /** Σ byte-allowance the n-ary interior passes those bridges. The allowance
176
+ * is `middleBytes + (m + 1) * W` — every intermediate answer's bytes plus
177
+ * one window of glue per joint — so it is the quantity that grows with a hub
178
+ * query's answers, and the first thing to read when the peak moves. */
179
+ coverAllowanceBytes: number;
155
180
  private readonly _phases;
156
181
  private readonly _t0;
157
182
  /** Every work counter's current value, by name — the snapshot `time`
package/dist/src/meter.js CHANGED
@@ -149,6 +149,50 @@ export class Meter {
149
149
  mechanismRuns = 0;
150
150
  /** Candidates the decider weighed. */
151
151
  candidates = 0;
152
+ // ── Graph search: the fact join (DIRECTION) ─────────────────────────────
153
+ //
154
+ // The join's outcome was observable ONLY through the rationale, and the
155
+ // rationale PERTURBS the search (measured: appending text to a refusal note
156
+ // changed a traced answer). These four counters are the untraced view — the
157
+ // same surface every other work counter uses, incremented where the decision
158
+ // is made, never behind a trace guard.
159
+ /** `deriveThrough` yielded — a fact was reached through the subject the query
160
+ * never named. */
161
+ joinFired = 0;
162
+ /** Refused: no key names the entity and the tail together. (A key that
163
+ * resolves but leads nowhere is not "refused" — it is not the relation, so
164
+ * the scan simply moves on; there is no counter for a case the loop cannot
165
+ * reach.) */
166
+ joinNoKey = 0;
167
+ /** Refused: the fact contains no entity that leads anywhere. */
168
+ joinNoEntity = 0;
169
+ // ── Mind: the multi-hop pivot (EXTENSION) ───────────────────────────────
170
+ //
171
+ // `pivotStep` was observable only through the rationale, and the rationale
172
+ // perturbs the search (measured). How far the reasoner hopped is a
173
+ // BEHAVIOUR, so it needs an untraced view: one counter, incremented where the
174
+ // step is emitted.
175
+ /** Times the reasoner pivoted on a span its answer contains and stepped
176
+ * across that fact. */
177
+ pivotSteps = 0;
178
+ // ── Mind: the cover's connector assembly (LIMIT) ────────────────────────
179
+ //
180
+ // The cover's `run` is 91% of a hub query's time (`"Hello."`: 2.7 s of 3.0 s)
181
+ // and holds its ~270 MB peak, and none of it was countable: `searchPushes`
182
+ // and `candidates` do not see the connector assembly. These two counters are
183
+ // the untraced view of it.
184
+ /** `bridge` calls the cover makes assembling connectors (pairwise + n-ary). */
185
+ coverBridges = 0;
186
+ /** Continuations a CHAIN hop offered the search. Bounded by the question
187
+ * (`ceil(queryLen / W)`) rather than by the corpus's fan-out — measured on a
188
+ * hub of degree 1083, offering every continuation grew the chart to 3113 outs
189
+ * and cost a 270 MB peak / 256 MB OOM for a two-word question. */
190
+ chainOffers = 0;
191
+ /** Σ byte-allowance the n-ary interior passes those bridges. The allowance
192
+ * is `middleBytes + (m + 1) * W` — every intermediate answer's bytes plus
193
+ * one window of glue per joint — so it is the quantity that grows with a hub
194
+ * query's answers, and the first thing to read when the peak moves. */
195
+ coverAllowanceBytes = 0;
152
196
  // ── Phases ──────────────────────────────────────────────────────────────
153
197
  _phases = new Map();
154
198
  _t0 = performance.now();
@@ -0,0 +1,40 @@
1
+ import type { MindContext } from "./types.js";
2
+ /** One stored experience pair, as bytes. */
3
+ export interface CorpusPair {
4
+ context: Uint8Array;
5
+ continuation: Uint8Array;
6
+ contextId: number;
7
+ continuationId: number;
8
+ /** Bytes of the query this pair was matched on — 0 when browsing. */
9
+ matchedBytes: number;
10
+ /** True when the stored bytes ran past the declared preview capacity. */
11
+ contextTruncated: boolean;
12
+ continuationTruncated: boolean;
13
+ }
14
+ /** Why a search produced no pairs. A STATE, so a caller's own layer can say it
15
+ * in its own words — the byte layer does not speak. */
16
+ export type CorpusMiss = "matched" | "nothing-resolved" | "no-continuations";
17
+ export interface CorpusResult {
18
+ pairs: CorpusPair[];
19
+ /** Query subtrees that content-addressed to a real stored node. */
20
+ resolved: number;
21
+ /** Distinct edge-bearing contexts the climb reached. */
22
+ reached: number;
23
+ /** Distinct contexts that carry a learnt continuation, store-wide. */
24
+ totalContexts: number;
25
+ /** True when these are browse samples rather than search results. */
26
+ browsed: boolean;
27
+ miss: CorpusMiss;
28
+ }
29
+ /** Which stored notes does this query reach? BYTES in, BYTES out.
30
+ *
31
+ * Exact content addressing through the machinery that already exists: the
32
+ * query's recognised sites are the resolved subtrees, the climb goes up from
33
+ * the biggest first, and a pair is a context that carries a continuation. */
34
+ export declare function searchCorpus(ctx: MindContext, queryBytes: Uint8Array, limit?: number): CorpusResult;
35
+ /** Browse real pairs, striding the id space so the sample is spread rather than
36
+ * one local cluster. DETERMINISTIC: `from` is the caller's own offset, so
37
+ * browsing twice with different offsets shows different notes without a random
38
+ * draw (the demo drew `Math.random()`, which the engine cannot do — same seed,
39
+ * same order, same query must mean the same answer). */
40
+ export declare function sampleCorpus(ctx: MindContext, limit?: number, from?: number): CorpusResult;
@@ -0,0 +1,149 @@
1
+ // corpus.ts — read the trained memory back out of the DAG, AS DATA.
2
+ //
3
+ // A trained experience pair IS one continuation edge: `src` is the context that
4
+ // was deposited, `dst` is what the mind learnt follows it. Reading them back
5
+ // uses the store's own structure and its own indexes — no auxiliary index is
6
+ // built, nothing is written, and NOTHING HERE KNOWS ABOUT TEXT: this layer takes
7
+ // bytes and returns bytes. The text case is one helper on the Mind
8
+ // (`searchCorpusText`), which encodes, calls this, and decodes.
9
+ //
10
+ // WHERE EACH STAGE COMES FROM (ported from the demo's `explore.ts`, which
11
+ // hand-rolled its own resolution):
12
+ //
13
+ // 1. PERCEIVE, CONTENT-ADDRESS and ADMIT the query — `recognise()`, the SAME
14
+ // machinery an answer goes through. It returns the sites: the query spans
15
+ // that content-addressed to a stored node that can lead somewhere. The
16
+ // demo's own recursive `findLeaf`/`findBranch` walk was a second
17
+ // implementation of exactly this, and it is NOT ported.
18
+ // 2. CLIMB the structural `kid` table from each resolved site to the
19
+ // edge-bearing contexts above it (`edgeAncestors`, traverse.ts), weighting
20
+ // each context by how much query content reached it.
21
+ // 3. READ the continuation off the edge table (`nextFirst`).
22
+ //
23
+ // COST is set by how much of the QUERY resolves, never by the size of the
24
+ // store: the sites are what recognition already found, the climb is bounded by
25
+ // the declared `corpusClimbs`/`corpusContextsPerClimb`, and every store call is
26
+ // a point probe or a capped read. All work is accounted by the store's own
27
+ // meter hooks when a response's meter is open — there is no second instrument.
28
+ //
29
+ // WHAT THIS IS NOT. Exact content addressing, not fuzzy keyword search: a query
30
+ // shares results with a stored note when it shares actual chunk-aligned content
31
+ // with it. An arbitrary mid-word fragment resolves to nothing, and the honest
32
+ // answer there is "nothing matched" — which is why `sampleCorpus` exists, and
33
+ // why the miss is reported as a STATE rather than as prose (the text helper
34
+ // turns it into words).
35
+ import { recognise } from "./recognition.js";
36
+ import { edgeAncestors } from "./traverse.js";
37
+ /** A context node as a pair, or null when it carries no continuation. */
38
+ function pairOf(ctx, id, matchedBytes) {
39
+ const outs = ctx.store.nextFirst(id, 1);
40
+ if (outs.length === 0)
41
+ return null;
42
+ const cap = ctx.cfg.corpusPreviewBytes;
43
+ const context = ctx.store.bytesPrefix(id, cap + 1);
44
+ const continuation = ctx.store.bytesPrefix(outs[0], cap + 1);
45
+ const contextTruncated = context.length > cap;
46
+ const continuationTruncated = continuation.length > cap;
47
+ return {
48
+ context: contextTruncated ? context.subarray(0, cap) : context,
49
+ continuation: continuationTruncated
50
+ ? continuation.subarray(0, cap)
51
+ : continuation,
52
+ contextId: id,
53
+ continuationId: outs[0],
54
+ matchedBytes,
55
+ contextTruncated,
56
+ continuationTruncated,
57
+ };
58
+ }
59
+ /** Which stored notes does this query reach? BYTES in, BYTES out.
60
+ *
61
+ * Exact content addressing through the machinery that already exists: the
62
+ * query's recognised sites are the resolved subtrees, the climb goes up from
63
+ * the biggest first, and a pair is a context that carries a continuation. */
64
+ export function searchCorpus(ctx, queryBytes, limit) {
65
+ const store = ctx.store;
66
+ const want = Math.max(1, Math.min(limit ?? ctx.cfg.corpusContextsPerClimb, ctx.cfg.corpusLimitMax));
67
+ // A resolved subtree must account for at least one window (W): a single
68
+ // character resolves against almost any store and means nothing. W is the
69
+ // mind's own line between chance and evidence — derived, never declared.
70
+ const floor = ctx.space.maxGroup;
71
+ const resolved = recognise(ctx, queryBytes).sites
72
+ .map((s) => ({ id: s.payload, len: store.contentLen(s.payload, 512) }))
73
+ .filter((r) => r.len >= floor);
74
+ // Biggest first, lowest id breaking ties: a clause is evidence, a character is
75
+ // noise, and equal evidence must not be decided by iteration order.
76
+ const byLength = resolved
77
+ .sort((a, b) => b.len - a.len || a.id - b.id)
78
+ .slice(0, ctx.cfg.corpusClimbs);
79
+ // Weight each context by how much query content reached it.
80
+ const weight = new Map();
81
+ for (const { id, len } of byLength) {
82
+ for (const root of edgeAncestors(ctx, id, ctx.cfg.corpusContextsPerClimb).roots) {
83
+ weight.set(root, (weight.get(root) ?? 0) + len);
84
+ }
85
+ }
86
+ const pairs = [];
87
+ const ranked = [...weight.entries()].sort((a, b) => b[1] - a[1] || a[0] - b[0]);
88
+ for (const [id, w] of ranked) {
89
+ if (pairs.length >= want)
90
+ break;
91
+ const pair = pairOf(ctx, id, w);
92
+ if (pair)
93
+ pairs.push(pair);
94
+ }
95
+ return {
96
+ pairs,
97
+ resolved: byLength.length,
98
+ reached: weight.size,
99
+ totalContexts: store.edgeSourceCount(),
100
+ browsed: false,
101
+ miss: pairs.length > 0
102
+ ? "matched"
103
+ : weight.size === 0
104
+ ? "nothing-resolved"
105
+ : "no-continuations",
106
+ };
107
+ }
108
+ /** Browse real pairs, striding the id space so the sample is spread rather than
109
+ * one local cluster. DETERMINISTIC: `from` is the caller's own offset, so
110
+ * browsing twice with different offsets shows different notes without a random
111
+ * draw (the demo drew `Math.random()`, which the engine cannot do — same seed,
112
+ * same order, same query must mean the same answer). */
113
+ export function sampleCorpus(ctx, limit, from = 0) {
114
+ const store = ctx.store;
115
+ const want = Math.max(1, Math.min(limit ?? ctx.cfg.corpusContextsPerClimb, ctx.cfg.corpusLimitMax));
116
+ const total = store.nodeCount();
117
+ const probes = ctx.cfg.corpusSampleProbes;
118
+ const floorBytes = ctx.cfg.corpusSampleFloorBytes;
119
+ const pairs = [];
120
+ // EACH CONTEXT AT MOST ONCE. Striding the id space revisits ids when the
121
+ // store is small relative to the probe budget (measured: a 160-node store
122
+ // returned the SAME pair six times for `limit: 6`), and a browse that repeats
123
+ // itself is not a browse. The demo had the same hole; it is invisible only on
124
+ // a store far larger than the probe budget.
125
+ const seen = new Set();
126
+ for (let i = 0; i < probes && pairs.length < want && total > 0; i++) {
127
+ const slot = (i / probes + from) % 1;
128
+ const id = Math.floor(slot * total);
129
+ if (seen.has(id))
130
+ continue;
131
+ if (!store.has(id) || !store.hasNext(id))
132
+ continue;
133
+ if (store.contentLen(id, floorBytes) < floorBytes)
134
+ continue;
135
+ const pair = pairOf(ctx, id, 0);
136
+ if (pair) {
137
+ seen.add(id);
138
+ pairs.push(pair);
139
+ }
140
+ }
141
+ return {
142
+ pairs,
143
+ resolved: 0,
144
+ reached: pairs.length,
145
+ totalContexts: store.edgeSourceCount(),
146
+ browsed: true,
147
+ miss: pairs.length > 0 ? "matched" : "no-continuations",
148
+ };
149
+ }
@@ -162,6 +162,13 @@ export declare class GraphSearch {
162
162
  * recursive completion), and chooseNext (distributional-evidence edge
163
163
  * disambiguation when a recognised form has multiple continuations). */
164
164
  host: GraphSearchHost);
165
+ /** The nodes the QUERY canonically names — the same identity the store's keys
166
+ * were written through. A byte-exact test is not enough: the query writes
167
+ * `Eiffel Tower country` and the deposited node is `eiffel tower country`, so
168
+ * a join that filters the query's own subject by RAW bytes re-admits it —
169
+ * measured: that is the trap's wrong answer (`The capital of Eiffel Tower
170
+ * country is Berlin.`). Cached by query identity, because the search is
171
+ * reused across responses. */
165
172
  private hubBound;
166
173
  /** Explore the Sema graph for the lightest cover of the query and return its
167
174
  * chosen spans left-to-right — WITH the derivation's total weight (the g