@hviana/sema 0.8.1 → 0.8.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (57) hide show
  1. package/dist/src/config.d.ts +17 -0
  2. package/dist/src/config.js +18 -0
  3. package/dist/src/meter.d.ts +25 -0
  4. package/dist/src/meter.js +44 -0
  5. package/dist/src/mind/corpus.d.ts +40 -0
  6. package/dist/src/mind/corpus.js +149 -0
  7. package/dist/src/mind/graph-search.d.ts +7 -0
  8. package/dist/src/mind/graph-search.js +235 -24
  9. package/dist/src/mind/index.d.ts +3 -1
  10. package/dist/src/mind/index.js +1 -0
  11. package/dist/src/mind/match.d.ts +8 -3
  12. package/dist/src/mind/match.js +142 -58
  13. package/dist/src/mind/mechanisms/cast.js +18 -2
  14. package/dist/src/mind/mechanisms/cover.js +6 -0
  15. package/dist/src/mind/mind.d.ts +55 -0
  16. package/dist/src/mind/mind.js +72 -2
  17. package/dist/src/mind/pipeline.js +25 -6
  18. package/dist/src/mind/reasoning.d.ts +5 -1
  19. package/dist/src/mind/reasoning.js +54 -1
  20. package/dist/src/mind/traverse.js +9 -1
  21. package/dist/src/mind/types.d.ts +22 -0
  22. package/docs/failures/tempting-but-wrong.md +31 -2
  23. package/jsr.json +1 -1
  24. package/package.json +1 -1
  25. package/src/config.ts +35 -0
  26. package/src/meter.ts +47 -0
  27. package/src/mind/corpus.ts +202 -0
  28. package/src/mind/graph-search.ts +252 -23
  29. package/src/mind/index.ts +8 -1
  30. package/src/mind/match.ts +143 -54
  31. package/src/mind/mechanisms/cast.ts +17 -1
  32. package/src/mind/mechanisms/cover.ts +5 -0
  33. package/src/mind/mind.ts +123 -0
  34. package/src/mind/pipeline.ts +30 -6
  35. package/src/mind/reasoning.ts +55 -0
  36. package/src/mind/traverse.ts +9 -1
  37. package/src/mind/types.ts +26 -0
  38. package/test/100-complete-grounding-trace.test.mjs +109 -0
  39. package/test/101-alignment-gap-bound.test.mjs +106 -0
  40. package/test/102-production-composes-at-scale.test.mjs +110 -0
  41. package/test/103-alignment-gap-budget.test.mjs +89 -0
  42. package/test/104-composition-is-reported.test.mjs +90 -0
  43. package/test/105-derive-through-reports-its-refusal.test.mjs +113 -0
  44. package/test/106-the-join-fires.test.mjs +94 -0
  45. package/test/107-the-join-is-counted.test.mjs +81 -0
  46. package/test/108-the-join-chains.test.mjs +78 -0
  47. package/test/109-the-pivot-is-counted.test.mjs +60 -0
  48. package/test/110-the-reasoner-stops-when-the-question-is-answered.test.mjs +91 -0
  49. package/test/111-the-cover-assembly-is-counted.test.mjs +74 -0
  50. package/test/112-the-exploration-does-not-grow-with-the-hub.test.mjs +89 -0
  51. package/test/113-the-rationale-payload-is-bounded.test.mjs +84 -0
  52. package/test/114-alignment-budget-is-per-sweep.test.mjs +93 -0
  53. package/test/116-the-extension-is-gated-by-the-pipelines-own-remainder.test.mjs +100 -0
  54. package/test/117-corpus-search.test.mjs +171 -0
  55. package/test/14-scaling.test.mjs +10 -7
  56. package/test/76-reference-binding.test.mjs +6 -1
  57. package/test/89-completion-recursion.test.mjs +30 -5
@@ -16,7 +16,7 @@ import { analogyStrength, follow, project, reverseContext, sharedFrameStrengthOf
16
16
  import { joinWithBridge } from "../resonance.js";
17
17
  import { restatesQuery } from "../reasoning.js";
18
18
  import { CONCEPT, STEP } from "../graph-search.js";
19
- import { concat2, indexOf } from "../../bytes.js";
19
+ import { indexOf } from "../../bytes.js";
20
20
  import { consensusFloor, dominates } from "../../geometry.js";
21
21
  import { unexplainedLabel, unexplainedSpans, } from "../rationale.js";
22
22
  import { rItem, rNode } from "../trace.js";
@@ -523,7 +523,23 @@ export async function counterfactualTransfer(ctx, query, pre) {
523
523
  const fwd = await follow(ctx, proj.anchor, qv);
524
524
  if (fwd !== null && indexOf(answer, fwd, 0) < 0 &&
525
525
  !restatesQuery(query, fwd)) {
526
- answer = concat2(answer, fwd);
526
+ // THROUGH THE SHARED JOINER, not a bare concatenation.
527
+ //
528
+ // `joinWithBridge` is the composition step every out-of-search assembly
529
+ // shares (multi-topic fusion, CAST's substitution and comparison): it
530
+ // asks the corpus for a learnt connector between the pieces and, on a
531
+ // miss, joins them BARE **and says so** — the `bridgeMiss` step (see
532
+ // resonance.ts). This site bypassed it, and that is the whole of the
533
+ // gluing the study measured: `"Steel is hard"` + `"wet"` came back as
534
+ // `"hardwet"`, `"eva director father"` + `"The father of…"` as
535
+ // `"fatherThe"` — compositions no rationale could show, because the one
536
+ // step that made them left no trace.
537
+ //
538
+ // Routing it through the shared joiner is the instrumentation fix that
539
+ // comes first: a bare join stays possible (the house rule is "joined
540
+ // bare, never silent") but it is now VISIBLE, and an attested connector
541
+ // is used when the corpus has one.
542
+ answer = await joinWithBridge(ctx, answer, fwd);
527
543
  }
528
544
  ctx.trace?.step("projectCounterfactual", [
529
545
  rItem(filler, "filler", subj.point.anchor),
@@ -83,6 +83,8 @@ export async function resolveConnectors(ctx, sites, query) {
83
83
  const bridgePair = async (l, r) => {
84
84
  if (l === r || links.has(l + "," + r))
85
85
  return;
86
+ if (ctx.meter)
87
+ ctx.meter.coverBridges++;
86
88
  const link = await bridge(ctx, read(ctx, l), read(ctx, r));
87
89
  if (link !== null)
88
90
  links.set(l + "," + r, link);
@@ -122,6 +124,10 @@ export async function resolveConnectors(ctx, sites, query) {
122
124
  // plus one W-quantum of glue per joint — pass that allowance so the
123
125
  // bridge's phrase-scale cap admits the whole learnt run.
124
126
  const allowance = middleBytes + (m + 1) * W;
127
+ if (ctx.meter) {
128
+ ctx.meter.coverBridges++;
129
+ ctx.meter.coverAllowanceBytes += allowance;
130
+ }
125
131
  const interior = await bridge(ctx, first.bytes, orderedNodes[m].bytes, allowance);
126
132
  if (interior !== null)
127
133
  links.set(key, interior);
@@ -1,5 +1,6 @@
1
1
  import { Vec } from "../vec.js";
2
2
  import { Sema, Space } from "../sema.js";
3
+ import type { CorpusResult } from "./corpus.js";
3
4
  import { Alphabet } from "../alphabet.js";
4
5
  import { Grid } from "../geometry.js";
5
6
  import { BoundedMap, type Store } from "../store.js";
@@ -49,10 +50,40 @@ import type { AttentionRead, MindContext, Recognition } from "./types.js";
49
50
  export type { AnchorRejectionReason, ClimbConsensusData, ConsensusAnchorTrace, ConsensusReachTrace, ConsensusRegionTrace, CrossRegionTier, JunctionVoteTrace, RegionOutcome, } from "./attention.js";
50
51
  export type { AncestorReach, SaturationReason, SaturationStop, } from "./types.js";
51
52
  import { type CostReport, Meter } from "../meter.js";
53
+ /** A stored pair as TEXT — the text helper's view of {@link CorpusPair}. */
54
+ export interface CorpusTextPair {
55
+ context: string;
56
+ continuation: string;
57
+ contextId: number;
58
+ continuationId: number;
59
+ matchedBytes: number;
60
+ contextTruncated: boolean;
61
+ continuationTruncated: boolean;
62
+ }
63
+ /** {@link CorpusResult} as text, plus the prose for why nothing matched. The
64
+ * byte layer reports a STATE; saying it in words belongs to the text layer. */
65
+ export interface CorpusTextResult {
66
+ query: string;
67
+ pairs: CorpusTextPair[];
68
+ resolved: number;
69
+ reached: number;
70
+ totalContexts: number;
71
+ browsed: boolean;
72
+ note?: string;
73
+ }
52
74
  export interface MindOptions {
53
75
  seed?: number;
54
76
  recallQueryK?: number;
55
77
  haloQueryK?: number;
78
+ /** Items one rationale step may itemise — see {@link MindConfig}. */
79
+ rationaleSampleK?: number;
80
+ /** Corpus-reading capacities and budgets — see {@link MindConfig}. */
81
+ corpusLimitMax?: number;
82
+ corpusClimbs?: number;
83
+ corpusContextsPerClimb?: number;
84
+ corpusSampleProbes?: number;
85
+ corpusPreviewBytes?: number;
86
+ corpusSampleFloorBytes?: number;
56
87
  normalizeEpsilon?: number;
57
88
  cosineEpsilon?: number;
58
89
  geometry?: Partial<import("../config.js").GeometryConfig>;
@@ -153,6 +184,10 @@ export declare class Mind implements MindContext {
153
184
  * `traverse.ts`'s ONE definition (edge or halo, with its response-scoped
154
185
  * cache). The search holds a bare Store and cannot reach that cache itself,
155
186
  * so it asks through this hook; a bare host keeps its raw-store fallback. */
187
+ /** The canonical identity for the search (see GraphSearchHost). */
188
+ canonResolve(bytes: Uint8Array): number | null;
189
+ /** Feed a search refusal into the rationale (see GraphSearchHost). */
190
+ reportSearch(name: string, parts: ReadonlyArray<Uint8Array>, note: string): void;
156
191
  leadsSomewhere(id: number): boolean;
157
192
  recogniseSpan(bytes: Uint8Array): {
158
193
  sites: ReadonlyArray<Site>;
@@ -167,6 +202,8 @@ export declare class Mind implements MindContext {
167
202
  * with the most distributional evidence (highest `prevOf` count — the
168
203
  * structural manifestation of its halo). When evidence is equal the
169
204
  * first-inserted edge wins. */
205
+ /** See {@link GraphSearchHost.contentCuts}. */
206
+ contentCuts(bytes: Uint8Array): readonly number[];
170
207
  chooseNext(node: number): number | undefined;
171
208
  constructor(opts?: MindOptions);
172
209
  constructor(cfg: MindConfig, store: Store, _fromStore: true);
@@ -249,6 +286,24 @@ export declare class Mind implements MindContext {
249
286
  * as one form, provided the store's canon index is built
250
287
  * ({@link buildCanonIndex}). */
251
288
  respondText(input: string, inspectRationale?: InspectRationale): Promise<string>;
289
+ /** Which stored notes does this query REACH? BYTES in, BYTES out — this
290
+ * method has no notion of text or encoding; the text case is
291
+ * {@link searchCorpusText}, which is one caller of this.
292
+ *
293
+ * Exact content addressing through the machinery an answer already uses
294
+ * (see src/mind/corpus.ts): the query's recognised sites are the resolved
295
+ * subtrees, the climb goes up from the biggest, and a result is a context
296
+ * that carries a learnt continuation. Nothing is written and nothing is
297
+ * indexed. */
298
+ searchCorpus(queryBytes: Uint8Array, limit?: number): CorpusResult;
299
+ /** Browse real pairs. Deterministic: `from` is the caller's own offset in
300
+ * [0,1), so browsing twice with different offsets shows different notes
301
+ * without a random draw. */
302
+ sampleCorpus(limit?: number, from?: number): CorpusResult;
303
+ /** The TEXT case of {@link searchCorpus}: encode, search, decode. The search
304
+ * itself exists once, in the byte layer above; only the rendering lives
305
+ * here, with the rest of this class's text modality. */
306
+ searchCorpusText(query: string, limit?: number): CorpusTextResult;
252
307
  /** Begin a new conversation, optionally restoring from a previously-saved
253
308
  * {@link ConversationState}. The returned handle is required for
254
309
  * {@link respondTurn} and {@link endConversation}.
@@ -9,8 +9,9 @@
9
9
  // Architecture: 4 primitives × 2 patterns = all inference.
10
10
  // Implementation split across src/mind/*.ts — this file assembles the Mind class.
11
11
  import { makeKeyring, rng, setVecConfig } from "../vec.js";
12
+ import { sampleCorpus, searchCorpus } from "./corpus.js";
12
13
  import { Alphabet } from "../alphabet.js";
13
- import { contentFoldIncremental, reachThreshold, } from "../geometry.js";
14
+ import { contentBoundaries, contentFoldIncremental, reachThreshold, } from "../geometry.js";
14
15
  import { BoundedMap } from "../store.js";
15
16
  import { SQliteStore } from "../store-sqlite.js";
16
17
  import { resolveConfig } from "../config.js";
@@ -19,7 +20,7 @@ import { bytesEqual, concat2 } from "../bytes.js";
19
20
  import { GraphSearch, } from "./graph-search.js";
20
21
  import { Alu } from "../alu/src/index.js";
21
22
  import { decodeText, Rationale, } from "./rationale.js";
22
- import { gistOf, inputBytes, perceive as perceiveImpl, perceiveKey, resolve as resolveImpl, } from "./primitives.js";
23
+ import { canonResolve as canonResolveImpl, gistOf, inputBytes, perceive as perceiveImpl, perceiveKey, resolve as resolveImpl, } from "./primitives.js";
23
24
  import { chooseNext, edgeAncestors as edgeAncestorsFn, invalidateStructuralCaches, leadsSomewhere, } from "./traverse.js";
24
25
  import { invalidateJunctionCache } from "./junction.js";
25
26
  import { follow } from "./match.js";
@@ -33,6 +34,21 @@ import { rItem } from "./trace.js";
33
34
  // The work meter is exported from src/index.ts (via src/meter.ts) — the one
34
35
  // definition; the Mind only consumes it.
35
36
  import { Meter } from "../meter.js";
37
+ /** What the text helper says when the byte layer reports a miss. */
38
+ const CORPUS_NOTE = {
39
+ "nothing-resolved": "No trained note sits above the parts of that text the mind recognised. " +
40
+ "It addresses content exactly, so try wording closer to something it was " +
41
+ "actually given — or browse the examples instead.",
42
+ "no-continuations": "That text reaches stored nodes, but none of them carries a learnt " +
43
+ "continuation.",
44
+ };
45
+ /** UTF-8 of bytes for display: reuse {@link decodeText} (the mind's own text
46
+ * conversion), then drop the replacement character a byte-boundary cut leaves
47
+ * behind. Much of a real corpus is non-Latin, so that trailing U+FFFD is the
48
+ * common case, not an exotic one — and it is the ONLY thing added here. */
49
+ function previewCorpusText(bytes) {
50
+ return decodeText(bytes).replace(/\uFFFD+$/, "").replace(/\s+/g, " ").trim();
51
+ }
36
52
  // ═══════════════════════════════════════════════════════════════════════════
37
53
  // THE MIND
38
54
  // ═══════════════════════════════════════════════════════════════════════════
@@ -116,6 +132,14 @@ export class Mind {
116
132
  * `traverse.ts`'s ONE definition (edge or halo, with its response-scoped
117
133
  * cache). The search holds a bare Store and cannot reach that cache itself,
118
134
  * so it asks through this hook; a bare host keeps its raw-store fallback. */
135
+ /** The canonical identity for the search (see GraphSearchHost). */
136
+ canonResolve(bytes) {
137
+ return canonResolveImpl(this, bytes);
138
+ }
139
+ /** Feed a search refusal into the rationale (see GraphSearchHost). */
140
+ reportSearch(name, parts, note) {
141
+ this.trace?.step(name, parts.map((b) => rItem(b)), [], note);
142
+ }
119
143
  leadsSomewhere(id) {
120
144
  return leadsSomewhere(this, id);
121
145
  }
@@ -136,6 +160,10 @@ export class Mind {
136
160
  * with the most distributional evidence (highest `prevOf` count — the
137
161
  * structural manifestation of its halo). When evidence is equal the
138
162
  * first-inserted edge wins. */
163
+ /** See {@link GraphSearchHost.contentCuts}. */
164
+ contentCuts(bytes) {
165
+ return contentBoundaries(this.space, bytes);
166
+ }
139
167
  chooseNext(node) {
140
168
  return chooseNext(this, node, this._edgeGuide);
141
169
  }
@@ -411,6 +439,48 @@ export class Mind {
411
439
  const r = await this.respond(input, inspectRationale);
412
440
  return decodeText(r.bytes);
413
441
  }
442
+ // ── Reading the trained memory back ─────────────────────────────────────
443
+ /** Which stored notes does this query REACH? BYTES in, BYTES out — this
444
+ * method has no notion of text or encoding; the text case is
445
+ * {@link searchCorpusText}, which is one caller of this.
446
+ *
447
+ * Exact content addressing through the machinery an answer already uses
448
+ * (see src/mind/corpus.ts): the query's recognised sites are the resolved
449
+ * subtrees, the climb goes up from the biggest, and a result is a context
450
+ * that carries a learnt continuation. Nothing is written and nothing is
451
+ * indexed. */
452
+ searchCorpus(queryBytes, limit) {
453
+ return searchCorpus(this, queryBytes, limit);
454
+ }
455
+ /** Browse real pairs. Deterministic: `from` is the caller's own offset in
456
+ * [0,1), so browsing twice with different offsets shows different notes
457
+ * without a random draw. */
458
+ sampleCorpus(limit, from) {
459
+ return sampleCorpus(this, limit, from);
460
+ }
461
+ /** The TEXT case of {@link searchCorpus}: encode, search, decode. The search
462
+ * itself exists once, in the byte layer above; only the rendering lives
463
+ * here, with the rest of this class's text modality. */
464
+ searchCorpusText(query, limit) {
465
+ const result = this.searchCorpus(new TextEncoder().encode(query), limit);
466
+ return {
467
+ query,
468
+ pairs: result.pairs.map((p) => ({
469
+ context: previewCorpusText(p.context),
470
+ continuation: previewCorpusText(p.continuation),
471
+ contextId: p.contextId,
472
+ continuationId: p.continuationId,
473
+ matchedBytes: p.matchedBytes,
474
+ contextTruncated: p.contextTruncated,
475
+ continuationTruncated: p.continuationTruncated,
476
+ })),
477
+ resolved: result.resolved,
478
+ reached: result.reached,
479
+ totalContexts: result.totalContexts,
480
+ browsed: result.browsed,
481
+ note: result.miss === "matched" ? undefined : CORPUS_NOTE[result.miss],
482
+ };
483
+ }
414
484
  // ── Conversation API ────────────────────────────────────────────────────
415
485
  /** Begin a new conversation, optionally restoring from a previously-saved
416
486
  * {@link ConversationState}. The returned handle is required for
@@ -355,9 +355,32 @@ export async function think(ctx, query, mechs) {
355
355
  const voiced = (provenance === "cast" || provenance === "join")
356
356
  ? [...castUsed].flatMap((id) => ctx.store.nextFirst(id, hubBound(ctx)).map((n) => read(ctx, n)))
357
357
  : [];
358
+ // REPORTABLE, NOT SILENT. A declared-complete grounding ends the derivation
359
+ // here, and that decision is part of the derivation's shape: the reader of a
360
+ // rationale must be able to see that the chain stopped because the mechanism
361
+ // claimed the query WAS the context, not because nothing followed. The step
362
+ // carries the claim, not a re-description of the answer — the extension is
363
+ // skipped, so there is no output item to show.
364
+ if (decided.complete) {
365
+ ctx.trace?.step("completeGrounding", [rItem(answer, provenance)], [], "grounding declared complete — the query IS the context, so " +
366
+ "post-grounding extension is skipped");
367
+ }
368
+ // THE REASONER JUDGES ITS OWN EXTENSIONS BY THE PIPELINE'S REMAINDER, not by
369
+ // the ladder's `accounted` — and by the SAME reading the fuse gate below uses,
370
+ // with the same W floor. `accounted` is a COST quantity (measured: a query
371
+ // fully explained by one computed span plus bridged connectors reports
372
+ // `accounted: []` while nothing is unexplained), and a remainder under one
373
+ // river-fold quantum is bridging punctuation, never a second topic — so it
374
+ // licenses no extension and blocks none.
375
+ const explained = [
376
+ ...decided.accounted,
377
+ ...pre.computed.map((u) => [u.i, u.j]),
378
+ ];
379
+ const uncovered = unexplainedSpans(query.length, explained)
380
+ .filter(([a, b]) => b - a >= ctx.space.maxGroup);
358
381
  const reasoned = decided.complete ? answer : meter
359
- ? await meter.time("reason", () => reason(ctx, query, answer, preConsumed, pre, voiced))
360
- : await reason(ctx, query, answer, preConsumed, pre, voiced);
382
+ ? await meter.time("reason", () => reason(ctx, query, answer, preConsumed, pre, voiced, uncovered))
383
+ : await reason(ctx, query, answer, preConsumed, pre, voiced, uncovered);
361
384
  // Fuse only when the query has a genuine REMAINDER no mechanism's
362
385
  // structural evidence touched at all. `decided.accounted` alone
363
386
  // undercounts this: it is a COST-LADDER quantity (cover.ts prices its
@@ -374,10 +397,6 @@ export async function think(ctx, query, mechs) {
374
397
  // observed: a single space between two fully-computed arithmetic spans
375
398
  // ("2+2 3+3") registered as "unaccounted" and pulled in an unrelated
376
399
  // corpus fact, corrupting "4 6" into "4 63".
377
- const explained = [
378
- ...decided.accounted,
379
- ...pre.computed.map((u) => [u.i, u.j]),
380
- ];
381
400
  const remainder = unaccounted(explained);
382
401
  // Whether the winning candidate's entire recognised substance is
383
402
  // COMPUTED — every accounted span exactly a pre.computed span, nothing
@@ -20,7 +20,11 @@ export declare function restatesQuery(query: Uint8Array, bytes: Uint8Array): boo
20
20
  * when it declared one — see the pivot's own containment rule. `pre` is the
21
21
  * response's shared pre-computation — the post-grounding stages read the
22
22
  * same container the mechanisms did. */
23
- export declare function reason(ctx: MindContext, query: Uint8Array, answer: Uint8Array, preConsumed: ReadonlySet<number>, pre: Precomputed, voiced?: readonly Uint8Array[]): Promise<Uint8Array>;
23
+ export declare function reason(ctx: MindContext, query: Uint8Array, answer: Uint8Array, preConsumed: ReadonlySet<number>, pre: Precomputed, voiced?: readonly Uint8Array[],
24
+ /** The query material the GROUNDING left uncovered — the cost ladder's own
25
+ * `unaccounted` spans. Only the reasoner's OWN extensions are judged
26
+ * against it; a mechanism carrying its own `used` set owns its shape. */
27
+ uncovered?: readonly (readonly [number, number])[]): Promise<Uint8Array>;
24
28
  /** Fuse independent points of attention into one answer (multi-topic).
25
29
  * When the consensus climb finds more than one dominant point, each
26
30
  * independent point grounds its own answer; they are bridged together
@@ -30,7 +30,11 @@ export function restatesQuery(query, bytes) {
30
30
  * when it declared one — see the pivot's own containment rule. `pre` is the
31
31
  * response's shared pre-computation — the post-grounding stages read the
32
32
  * same container the mechanisms did. */
33
- export async function reason(ctx, query, answer, preConsumed, pre, voiced = []) {
33
+ export async function reason(ctx, query, answer, preConsumed, pre, voiced = [],
34
+ /** The query material the GROUNDING left uncovered — the cost ladder's own
35
+ * `unaccounted` spans. Only the reasoner's OWN extensions are judged
36
+ * against it; a mechanism carrying its own `used` set owns its shape. */
37
+ uncovered = []) {
34
38
  // Echo guard: a query that is ITSELF a learnt continuation (some context's
35
39
  // answer) is being asked back at the system — hopping forward from it would
36
40
  // chain through the very fact that produced it and echo the conversation
@@ -178,6 +182,55 @@ export async function reason(ctx, query, answer, preConsumed, pre, voiced = [])
178
182
  consumeAll(pivot);
179
183
  if (fc === null || bytesEqual(fc, cur) || restatesQuery(query, fc))
180
184
  break;
185
+ // WHOSE EXTENSION IS THIS?
186
+ //
187
+ // `voiced` is what the mechanism WITHHELD (the pipeline sends the used
188
+ // anchors' CONTINUATIONS, not their bytes — see pipeline's own note), so a
189
+ // non-empty `voiced` means exactly what that note says: the grounding came
190
+ // from a mechanism that carries its own short `used` set (cast/join) and
191
+ // therefore owns the shape of its answer. The further terms inside such a
192
+ // seat are legitimately followable — test/29 C3's `Mona Lisa` lives inside
193
+ // the voiced seat and leads on to a fact about neither analog.
194
+ //
195
+ // Every other grounding is ordinary, and an extension of it is the
196
+ // reasoner's own inference: it is taken only while question material the
197
+ // grounding left uncovered remains AND the step carries some of it, judged
198
+ // by the mind's own line between chance and evidence — one W-byte window,
199
+ // no word notion, no character class, no threshold. Measured: the drift's
200
+ // second step (`the Eiffel Tower is in Paris` after `Paris is famous for
201
+ // the Eiffel Tower`) carries no window of `" famous for"` and is refused,
202
+ // while the first carries it. Terminates by a real argument: the uncovered
203
+ // material is finite and each taken extension must carry some of it.
204
+ const producerOwnsShape = voiced.length > 0;
205
+ if (!producerOwnsShape && uncovered.length > 0) {
206
+ const W = ctx.space.maxGroup;
207
+ let progress = false;
208
+ for (const [a, b] of uncovered) {
209
+ for (let i = a; i + W <= b && !progress; i++) {
210
+ if (indexOf(fc, query.subarray(i, i + W), 0) >= 0)
211
+ progress = true;
212
+ }
213
+ if (progress)
214
+ break;
215
+ }
216
+ if (!progress) {
217
+ // THE BRAKE, MADE VISIBLE. The reasoner declines a step that carries
218
+ // none of the material the grounding left uncovered — the drift the
219
+ // extension tests pin. A refusal that leaves no trace is the kind of
220
+ // silent cut AGENTS §6 forbids: the rationale is where a reader learns
221
+ // that an extension was declined for want of question material, and
222
+ // where the next person sees why the chain stopped here. Measured with
223
+ // the check disabled, test/110 and test/116 fail — so this brake is the
224
+ // only thing keeping the extension honest until the pivot reports its
225
+ // own accounted spans and the ladder can judge it instead.
226
+ const left = uncovered.reduce((n, [a, b]) => n + (b - a), 0);
227
+ ctx.trace?.step("pivotRefused", [rItem(cur, "answer"), rItem(query, "query")], uncovered.map(([a, b]) => rItem(query.subarray(a, b), "uncovered")), `the step carries none of the question material the grounding left ` +
228
+ `uncovered (${left} byte(s) in ${uncovered.length} span(s)) — refused`);
229
+ break;
230
+ }
231
+ }
232
+ if (ctx.meter)
233
+ ctx.meter.pivotSteps++;
181
234
  t ??= ctx.trace?.enter("reason", [rItem(startedFrom, "grounded")]);
182
235
  ctx.trace?.step("pivotStep", [rItem(cur, "answer"), rNode(ctx, pivot, "pivot")], [rItem(fc, "answer", resolve(ctx, fc) ?? undefined)], "pivot on the shared span this answer contains, then step forward across that fact");
183
236
  cur = fc;
@@ -664,7 +664,15 @@ export function chooseNext(ctx, id, guide) {
664
664
  // for the prevCount calls in the loop above, never for extra rItemShort
665
665
  // byte-reads.
666
666
  if (ctx.trace) {
667
- const others = capped.filter((c) => c !== best);
667
+ // A BOUNDED SAMPLE, AND THE COUNT. The step used to carry EVERY candidate
668
+ // it weighed — measured on the trained store, 1559 out-items in one step
669
+ // (hubBound's own size) and 1082 in another (the hub's degree). The
670
+ // rationale's job is to explain the CHOICE, and the count is what says how
671
+ // wide the field was; the declared candidate budget (`recallQueryK`) is what
672
+ // bounds the sample, so no number is invented here.
673
+ const others = capped
674
+ .filter((c) => c !== best)
675
+ .slice(0, ctx.cfg.rationaleSampleK);
668
676
  ctx.trace.step("disambiguate", [rItemShort(ctx, best, "halo-evidence", bestSupport)], others.map((c) => rItemShort(ctx, c, "candidate", ctx.store.prevCount(c))), `${capped.length} continuations — distributional evidence selects ` +
669
677
  `the most corroborated (distinct contexts ${bestSupport}, ` +
670
678
  `poured mass ${bestMass})`);
@@ -35,12 +35,34 @@ export interface GraphSearchHost {
35
35
  starts: ReadonlySet<number>;
36
36
  };
37
37
  chooseNext?(node: number): number | undefined;
38
+ /** The boundary positions of `bytes` under the engine's ONE boundary rule
39
+ * (geometry.ts's `contentBoundaries`), or undefined when the host has no
40
+ * space to ask. The join's key is an entity plus a prefix of the tail, and
41
+ * the prefix that names a stored relation ENDS on one of these boundaries —
42
+ * measured, 5 of 5 accepted keys over four join-firing queries, where the
43
+ * byte-by-byte scan spent 153 probes for 14 boundaries. Boundaries are
44
+ * content-defined and STABLE under prefix extension, which is why a corpus
45
+ * key's end is a boundary of the query's own fold of the same bytes. */
46
+ contentCuts?(bytes: Uint8Array): readonly number[];
38
47
  /** The admission predicate — `traverse.ts`'s `leadsSomewhere`, its ONE
39
48
  * definition: does this node bear an edge or a halo? Optional, so a bare
40
49
  * host (a raw Store and nothing else) still works; when present, the search
41
50
  * uses it rather than re-probing the store, which keeps the predicate
42
51
  * single-defined AND memoised on the response-scoped struct cache. */
43
52
  leadsSomewhere?(id: number): boolean;
53
+ /** Report a SEARCH REFUSAL into the rationale — the channel AGENTS §6
54
+ * requires: a callback threaded through a call chain must FEED the
55
+ * rationale, the way `GraphSearch`'s `onDerivation` feeds `traceDerivation`,
56
+ * never a channel of its own. Optional, so a bare host stays silent rather
57
+ * than crashing. */
58
+ reportSearch?(name: string, parts: ReadonlyArray<Uint8Array>, note: string): void;
59
+ /** The CANONICAL resolver ({@link canonResolve}), optional like
60
+ * {@link leadsSomewhere}. The store's keys were written through the
61
+ * canonical fold, so a fact's `Gustaf Molander` and the deposited
62
+ * `gustaf molander` are the SAME node (measured inside a response: the
63
+ * canonical resolver maps the surface form to the deposited node while a raw
64
+ * resolve returns null). A bare host falls back to the plain probe. */
65
+ canonResolve?(bytes: Uint8Array): number | null;
44
66
  }
45
67
  export interface Recognition {
46
68
  /** Forms that can lead somewhere — they have an edge or a halo. */
@@ -1,6 +1,6 @@
1
- # Tempting but Wrong — 12 Traps
1
+ # Tempting but Wrong — 13 Traps
2
2
 
3
- Twelve shortcuts that look plausible and break an invariant. Each states what
3
+ Thirteen shortcuts that look plausible and break an invariant. Each states what
4
4
  not to do, why it fails, and what to do instead.
5
5
 
6
6
  ### 1. `score >= threshold` decides identity
@@ -141,3 +141,32 @@ not to do, why it fails, and what to do instead.
141
141
  `twoEndedSeat`); turns are API state in `mind/mind.ts`, not segmentation.
142
142
  Pinned by `test/59-fold-invariance.test.mjs` and
143
143
  `test/63-fold-invariants.test.mjs`.
144
+
145
+ ### 13. Capping a combinatorial explosion instead of budgeting it
146
+
147
+ - **WRONG:** Answer a combinatorial explosion with a geometry-derived limit — a
148
+ cap on the pairs a sweep enumerates, the continuations a hop may offer, the
149
+ candidates a scan probes. A derived limit is the right cutoff for a DECISION;
150
+ used as the answer to explosion it is a short-circuit.
151
+ - **WHY:** It stops the computation silently. Reach is lost, the capability that
152
+ depended on it goes with it, and no test fails, because the tests were written
153
+ against the capped behaviour. Capping and removing the cap are both wrong:
154
+ capping truncates, removing lets the cost run, and the two failure modes hide
155
+ each other.
156
+ - **CORRECT:** BUDGET it. The work is charged in the one currency
157
+ (`MICRO`/`STEP`/`CONCEPT`/`PASS`; `weight = moves + PASS·unaccounted`, see
158
+ `docs/architecture/cost-model.md`), the charge is visible in the meter and the
159
+ rationale, and the SEARCH decides whether the work is worth paying — so
160
+ inference is never locked by a limit and nothing is truncated in silence.
161
+ Where the work is mechanical rather than evidential — enumeration, scans,
162
+ sweeps — the answer is an algorithm whose cost is structural in the bytes it
163
+ is given, not a smaller cap.
164
+ - **THE IDEAL:** a universal **closure engine** — one law of closure, stated in
165
+ the quantities the machine already has (`leadsSomewhere`, the
166
+ exact-then-canonical identity, `accounted` bytes, the ladder, `hubBound`),
167
+ from which the reach of a gap, the offer of a hop, the depth of a join and the
168
+ scope of a substitution are CONSEQUENCES, not four separate decisions. Nothing
169
+ in this repository is that today.
170
+ - **THE STANDARD A CHANGE MUST MEET:** state which consequence it is, and show
171
+ it following from the law. A change that cannot be stated that way is not
172
+ ready.
package/jsr.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "$schema": "https://jsr.io/schema/config-file.v1.json",
3
3
  "name": "@hviana/sema",
4
- "version": "0.8.1",
4
+ "version": "0.8.2",
5
5
  "exports": "./src/index.ts"
6
6
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@hviana/sema",
3
- "version": "0.8.1",
3
+ "version": "0.8.2",
4
4
  "description": "Sema: a non-parametric, instance-based reasoning system.",
5
5
  "repository": {
6
6
  "type": "git",
package/src/config.ts CHANGED
@@ -116,6 +116,23 @@ export interface MindConfig {
116
116
  seed: number;
117
117
  recallQueryK: number;
118
118
  haloQueryK: number;
119
+ /** Corpus reading (see src/mind/corpus.ts): results per call, resolved
120
+ * nodes climbed from, contexts requested per climb, probes used to stride
121
+ * the id space when browsing, bytes of each side a preview keeps, and the
122
+ * smallest deposited note browsing will show. Capacities and budgets only —
123
+ * the one material floor (a resolved node must account for W bytes) is
124
+ * derived from the geometry, not declared here. */
125
+ corpusLimitMax: number;
126
+ corpusClimbs: number;
127
+ corpusContextsPerClimb: number;
128
+ corpusSampleProbes: number;
129
+ corpusPreviewBytes: number;
130
+ corpusSampleFloorBytes: number;
131
+ /** Items one rationale step may ITEMISE (the whole field is still counted in
132
+ * the step's note). A capacity of the rationale, not of recall: sharing
133
+ * `recallQueryK` meant `new Mind({recallQueryK: 100000})` un-bounded the very
134
+ * payload the bound exists for (found by an adversarial review). */
135
+ rationaleSampleK: number;
119
136
  normalizeEpsilon: number;
120
137
  cosineEpsilon: number;
121
138
 
@@ -131,6 +148,13 @@ export const DEFAULT_CONFIG: MindConfig = {
131
148
  seed: 42,
132
149
  recallQueryK: 12,
133
150
  haloQueryK: 12,
151
+ rationaleSampleK: 12,
152
+ corpusLimitMax: 24,
153
+ corpusClimbs: 24,
154
+ corpusContextsPerClimb: 6,
155
+ corpusSampleProbes: 6000,
156
+ corpusPreviewBytes: 220,
157
+ corpusSampleFloorBytes: 12,
134
158
  normalizeEpsilon: 1e-12,
135
159
  cosineEpsilon: 1e-12,
136
160
  alu: {
@@ -172,6 +196,17 @@ export function resolveConfig(opts: Partial<MindConfig> = {}): MindConfig {
172
196
  seed: opts.seed ?? DEFAULT_CONFIG.seed,
173
197
  recallQueryK: opts.recallQueryK ?? DEFAULT_CONFIG.recallQueryK,
174
198
  haloQueryK: opts.haloQueryK ?? DEFAULT_CONFIG.haloQueryK,
199
+ rationaleSampleK: opts.rationaleSampleK ?? DEFAULT_CONFIG.rationaleSampleK,
200
+ corpusLimitMax: opts.corpusLimitMax ?? DEFAULT_CONFIG.corpusLimitMax,
201
+ corpusClimbs: opts.corpusClimbs ?? DEFAULT_CONFIG.corpusClimbs,
202
+ corpusContextsPerClimb: opts.corpusContextsPerClimb ??
203
+ DEFAULT_CONFIG.corpusContextsPerClimb,
204
+ corpusSampleProbes: opts.corpusSampleProbes ??
205
+ DEFAULT_CONFIG.corpusSampleProbes,
206
+ corpusPreviewBytes: opts.corpusPreviewBytes ??
207
+ DEFAULT_CONFIG.corpusPreviewBytes,
208
+ corpusSampleFloorBytes: opts.corpusSampleFloorBytes ??
209
+ DEFAULT_CONFIG.corpusSampleFloorBytes,
175
210
  normalizeEpsilon: opts.normalizeEpsilon ?? DEFAULT_CONFIG.normalizeEpsilon,
176
211
  cosineEpsilon: opts.cosineEpsilon ?? DEFAULT_CONFIG.cosineEpsilon,
177
212
  alu: {
package/src/meter.ts CHANGED
@@ -206,6 +206,53 @@ export class Meter {
206
206
  /** Candidates the decider weighed. */
207
207
  candidates = 0;
208
208
 
209
+ // ── Graph search: the fact join (DIRECTION) ─────────────────────────────
210
+ //
211
+ // The join's outcome was observable ONLY through the rationale, and the
212
+ // rationale PERTURBS the search (measured: appending text to a refusal note
213
+ // changed a traced answer). These four counters are the untraced view — the
214
+ // same surface every other work counter uses, incremented where the decision
215
+ // is made, never behind a trace guard.
216
+ /** `deriveThrough` yielded — a fact was reached through the subject the query
217
+ * never named. */
218
+ joinFired = 0;
219
+ /** Refused: no key names the entity and the tail together. (A key that
220
+ * resolves but leads nowhere is not "refused" — it is not the relation, so
221
+ * the scan simply moves on; there is no counter for a case the loop cannot
222
+ * reach.) */
223
+ joinNoKey = 0;
224
+ /** Refused: the fact contains no entity that leads anywhere. */
225
+ joinNoEntity = 0;
226
+
227
+ // ── Mind: the multi-hop pivot (EXTENSION) ───────────────────────────────
228
+ //
229
+ // `pivotStep` was observable only through the rationale, and the rationale
230
+ // perturbs the search (measured). How far the reasoner hopped is a
231
+ // BEHAVIOUR, so it needs an untraced view: one counter, incremented where the
232
+ // step is emitted.
233
+ /** Times the reasoner pivoted on a span its answer contains and stepped
234
+ * across that fact. */
235
+ pivotSteps = 0;
236
+
237
+ // ── Mind: the cover's connector assembly (LIMIT) ────────────────────────
238
+ //
239
+ // The cover's `run` is 91% of a hub query's time (`"Hello."`: 2.7 s of 3.0 s)
240
+ // and holds its ~270 MB peak, and none of it was countable: `searchPushes`
241
+ // and `candidates` do not see the connector assembly. These two counters are
242
+ // the untraced view of it.
243
+ /** `bridge` calls the cover makes assembling connectors (pairwise + n-ary). */
244
+ coverBridges = 0;
245
+ /** Continuations a CHAIN hop offered the search. Bounded by the question
246
+ * (`ceil(queryLen / W)`) rather than by the corpus's fan-out — measured on a
247
+ * hub of degree 1083, offering every continuation grew the chart to 3113 outs
248
+ * and cost a 270 MB peak / 256 MB OOM for a two-word question. */
249
+ chainOffers = 0;
250
+ /** Σ byte-allowance the n-ary interior passes those bridges. The allowance
251
+ * is `middleBytes + (m + 1) * W` — every intermediate answer's bytes plus
252
+ * one window of glue per joint — so it is the quantity that grows with a hub
253
+ * query's answers, and the first thing to read when the peak moves. */
254
+ coverAllowanceBytes = 0;
255
+
209
256
  // ── Phases ──────────────────────────────────────────────────────────────
210
257
 
211
258
  private readonly _phases = new Map<string, PhaseCost>();