@hviana/sema 0.8.0 → 0.8.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +22 -1
- package/DATASETS.md +1 -1
- package/dist/example/train_base/config.js +2 -2
- package/dist/example/train_base/corpora/massive.js +1 -1
- package/dist/example/train_base/readers.js +1 -1
- package/dist/src/config.d.ts +17 -0
- package/dist/src/config.js +18 -0
- package/dist/src/geometry.d.ts +10 -10
- package/dist/src/geometry.js +25 -24
- package/dist/src/meter.d.ts +29 -12
- package/dist/src/meter.js +58 -14
- package/dist/src/mind/attention.js +12 -12
- package/dist/src/mind/bridge.d.ts +8 -8
- package/dist/src/mind/bridge.js +33 -32
- package/dist/src/mind/corpus.d.ts +40 -0
- package/dist/src/mind/corpus.js +149 -0
- package/dist/src/mind/graph-search.d.ts +7 -8
- package/dist/src/mind/graph-search.js +244 -32
- package/dist/src/mind/index.d.ts +3 -1
- package/dist/src/mind/index.js +1 -0
- package/dist/src/mind/junction.d.ts +1 -1
- package/dist/src/mind/junction.js +8 -8
- package/dist/src/mind/learning.js +36 -35
- package/dist/src/mind/match.d.ts +8 -3
- package/dist/src/mind/match.js +156 -71
- package/dist/src/mind/mechanisms/cast.js +18 -2
- package/dist/src/mind/mechanisms/cover.js +19 -12
- package/dist/src/mind/mechanisms/prefix-completion.js +24 -24
- package/dist/src/mind/mechanisms/recall.js +38 -40
- package/dist/src/mind/mechanisms/reference.js +16 -16
- package/dist/src/mind/mind.d.ts +61 -7
- package/dist/src/mind/mind.js +72 -2
- package/dist/src/mind/pipeline-mechanism.d.ts +10 -8
- package/dist/src/mind/pipeline-mechanism.js +25 -21
- package/dist/src/mind/pipeline.d.ts +9 -9
- package/dist/src/mind/pipeline.js +49 -29
- package/dist/src/mind/primitives.d.ts +5 -5
- package/dist/src/mind/primitives.js +5 -5
- package/dist/src/mind/reasoning.d.ts +5 -1
- package/dist/src/mind/reasoning.js +54 -1
- package/dist/src/mind/recognition.d.ts +14 -13
- package/dist/src/mind/recognition.js +23 -23
- package/dist/src/mind/resonance.js +21 -21
- package/dist/src/mind/traverse.d.ts +54 -52
- package/dist/src/mind/traverse.js +83 -73
- package/dist/src/mind/types.d.ts +26 -4
- package/dist/src/store.d.ts +12 -12
- package/dist/src/store.js +12 -12
- package/docs/INDEX.md +2 -2
- package/docs/architecture/exact-vs-approximate.md +2 -1
- package/docs/architecture/fold-contract.md +1 -1
- package/docs/failures/tempting-but-wrong.md +33 -5
- package/docs/harness/gates.md +7 -7
- package/example/train_base/config.ts +2 -2
- package/example/train_base/corpora/massive.ts +1 -1
- package/example/train_base/readers.ts +1 -1
- package/jsr.json +1 -1
- package/package.json +1 -1
- package/src/config.ts +35 -0
- package/src/geometry.ts +25 -24
- package/src/meter.ts +61 -14
- package/src/mind/attention.ts +12 -12
- package/src/mind/bridge.ts +33 -32
- package/src/mind/corpus.ts +202 -0
- package/src/mind/graph-search.ts +261 -31
- package/src/mind/index.ts +8 -1
- package/src/mind/junction.ts +8 -8
- package/src/mind/learning.ts +36 -35
- package/src/mind/match.ts +163 -73
- package/src/mind/mechanisms/cast.ts +17 -1
- package/src/mind/mechanisms/cover.ts +18 -12
- package/src/mind/mechanisms/prefix-completion.ts +24 -24
- package/src/mind/mechanisms/recall.ts +38 -40
- package/src/mind/mechanisms/reference.ts +16 -16
- package/src/mind/mind.ts +129 -7
- package/src/mind/pipeline-mechanism.ts +25 -21
- package/src/mind/pipeline.ts +63 -38
- package/src/mind/primitives.ts +5 -5
- package/src/mind/reasoning.ts +55 -0
- package/src/mind/recognition.ts +23 -23
- package/src/mind/resonance.ts +21 -21
- package/src/mind/traverse.ts +83 -73
- package/src/mind/types.ts +30 -4
- package/src/store.ts +20 -20
- package/test/08-storage.test.mjs +1 -1
- package/test/100-complete-grounding-trace.test.mjs +109 -0
- package/test/101-alignment-gap-bound.test.mjs +106 -0
- package/test/102-production-composes-at-scale.test.mjs +110 -0
- package/test/103-alignment-gap-budget.test.mjs +89 -0
- package/test/104-composition-is-reported.test.mjs +90 -0
- package/test/105-derive-through-reports-its-refusal.test.mjs +113 -0
- package/test/106-the-join-fires.test.mjs +94 -0
- package/test/107-the-join-is-counted.test.mjs +81 -0
- package/test/108-the-join-chains.test.mjs +78 -0
- package/test/109-the-pivot-is-counted.test.mjs +60 -0
- package/test/110-the-reasoner-stops-when-the-question-is-answered.test.mjs +91 -0
- package/test/111-the-cover-assembly-is-counted.test.mjs +74 -0
- package/test/112-the-exploration-does-not-grow-with-the-hub.test.mjs +89 -0
- package/test/113-the-rationale-payload-is-bounded.test.mjs +84 -0
- package/test/114-alignment-budget-is-per-sweep.test.mjs +93 -0
- package/test/116-the-extension-is-gated-by-the-pipelines-own-remainder.test.mjs +100 -0
- package/test/117-corpus-search.test.mjs +171 -0
- package/test/14-scaling.test.mjs +10 -7
- package/test/35-prefix-edge.test.mjs +1 -1
- package/test/40-choosenext-scale-guard.test.mjs +16 -17
- package/test/56-bridge-identity-admission.test.mjs +6 -6
- package/test/70-prefix-completion.test.mjs +4 -3
- package/test/72-prefix-candidate-supply.test.mjs +3 -3
- package/test/73-scaffolding-only-bridge-abstains.test.mjs +6 -6
- package/test/75-multiturn-context-optimisation.test.mjs +5 -5
- package/test/76-reference-binding.test.mjs +6 -1
- package/test/84-composed-answer-honesty.test.mjs +5 -6
- package/test/88-dependency-footprint.test.mjs +1 -1
- package/test/89-completion-recursion.test.mjs +47 -19
- package/test/90-connector-read-cap.test.mjs +10 -8
- package/test/93-regime-prediction.test.mjs +10 -10
- package/test/94-cross-region-budget.test.mjs +2 -2
- package/test/95-wide-resonance-removed.test.mjs +8 -7
- package/test/96-bytes-walk-termination.test.mjs +3 -3
|
@@ -2,8 +2,8 @@
|
|
|
2
2
|
// bytes (Grounding IV).
|
|
3
3
|
//
|
|
4
4
|
// This file is a CONFIGURATION of the shared frame reading in match.ts, not a
|
|
5
|
-
// pipeline of its own.
|
|
6
|
-
// them and are reachable by any mechanism:
|
|
5
|
+
// pipeline of its own. The three parts it configures live where
|
|
6
|
+
// match-project.md puts them and are reachable by any mechanism:
|
|
7
7
|
//
|
|
8
8
|
// matcher Precomputed.frames() — the frame INVENTORY: which ranked
|
|
9
9
|
// candidates read as instances of the query's own frame, and
|
|
@@ -29,10 +29,10 @@
|
|
|
29
29
|
// candidate's continuation UNSUBSTITUTED, so admitting a slot-gap there would
|
|
30
30
|
// voice the corpus's filler for the asker's referent — the misreference
|
|
31
31
|
// measured live on the trained store ("How do you say 'flurbish' in French?"
|
|
32
|
-
// answered "the way to say hello is \"Bonjour\"").
|
|
32
|
+
// answered "the way to say hello is \"Bonjour\""). Nor is CAST rewired: its
|
|
33
33
|
// frame gate is WEAVE-local while a slot is COHORT-local, and substituting one
|
|
34
|
-
// population for the other is the error
|
|
35
|
-
// AVAILABLE, never imposed.
|
|
34
|
+
// population for the other is the error commonality.md names. The notion is
|
|
35
|
+
// made AVAILABLE, never imposed.
|
|
36
36
|
|
|
37
37
|
import type { MindContext } from "../types.js";
|
|
38
38
|
import type { FrameInstance } from "../match.js";
|
|
@@ -52,16 +52,16 @@ import { rItem, rNode, traceFail } from "../trace.js";
|
|
|
52
52
|
* agrees with nothing, so no carriage is attested — the same "two or no
|
|
53
53
|
* constituent" reading frame-filler's contentRuns applies.
|
|
54
54
|
*
|
|
55
|
-
* THIS IS ALSO THE MECHANISM'S REACH.
|
|
56
|
-
*
|
|
57
|
-
*
|
|
58
|
-
*
|
|
59
|
-
*
|
|
60
|
-
*
|
|
61
|
-
*
|
|
62
|
-
*
|
|
63
|
-
*
|
|
64
|
-
*
|
|
55
|
+
* THIS IS ALSO THE MECHANISM'S REACH. Evidence comes from the shared top-k
|
|
56
|
+
* resonance, so a frame the corpus instantiates only ONCE within k is not
|
|
57
|
+
* reachable here. Measured on the trained store: `How do you say 'flurbish' in
|
|
58
|
+
* French?` finds one instance of its frame in the top 24 — the rest are `How do
|
|
59
|
+
* you make …`, a different frame — so this abstains and recall's
|
|
60
|
+
* scaffolding-dominated tier answers with the CORPUS's filler. That
|
|
61
|
+
* misreference is recall's, and widening the supply is not the fix: the
|
|
62
|
+
* exhaustive √N list recall's refusal path builds costs hundreds of
|
|
63
|
+
* milliseconds and this runs before it. Abstaining on thin evidence is the
|
|
64
|
+
* honest reading (INVARIANTS.md). */
|
|
65
65
|
const MIN_INSTANCES = 2;
|
|
66
66
|
|
|
67
67
|
/** THE VOICING GATES — this mechanism's own reading of a pairing, applied here
|
|
@@ -135,7 +135,7 @@ function electFrame(
|
|
|
135
135
|
let best: FrameInstance[] = [];
|
|
136
136
|
for (const group of bySignature.values()) {
|
|
137
137
|
// Ties keep the FIRST group in insertion order, which is resonance rank —
|
|
138
|
-
// corpus-determined, like every other tie-break here (
|
|
138
|
+
// corpus-determined, like every other tie-break here (determinism.md).
|
|
139
139
|
if (group.length > best.length) best = group;
|
|
140
140
|
}
|
|
141
141
|
return best;
|
package/src/mind/mind.ts
CHANGED
|
@@ -11,9 +11,12 @@
|
|
|
11
11
|
|
|
12
12
|
import { cosine, makeKeyring, rng, setVecConfig, Vec } from "../vec.js";
|
|
13
13
|
import { bindSeat, fold, Sema, Space } from "../sema.js";
|
|
14
|
+
import { sampleCorpus, searchCorpus } from "./corpus.js";
|
|
15
|
+
import type { CorpusPair, CorpusResult } from "./corpus.js";
|
|
14
16
|
import { Alphabet } from "../alphabet.js";
|
|
15
17
|
import {
|
|
16
18
|
bytesToTree,
|
|
19
|
+
contentBoundaries,
|
|
17
20
|
contentFoldIncremental,
|
|
18
21
|
Grid,
|
|
19
22
|
gridToTree,
|
|
@@ -146,6 +149,7 @@ interface ConversationData {
|
|
|
146
149
|
import type { AttentionRead, MindContext, Recognition } from "./types.js";
|
|
147
150
|
import { changedNodes, liftAnswer, spliceAll } from "./types.js";
|
|
148
151
|
import {
|
|
152
|
+
canonResolve as canonResolveImpl,
|
|
149
153
|
foldTree,
|
|
150
154
|
gistOf,
|
|
151
155
|
inputBytes,
|
|
@@ -194,10 +198,61 @@ import { type CostReport, Meter } from "../meter.js";
|
|
|
194
198
|
|
|
195
199
|
// ── MindOptions ───────────────────────────────────────────────────────────
|
|
196
200
|
|
|
201
|
+
/** A stored pair as TEXT — the text helper's view of {@link CorpusPair}. */
|
|
202
|
+
export interface CorpusTextPair {
|
|
203
|
+
context: string;
|
|
204
|
+
continuation: string;
|
|
205
|
+
contextId: number;
|
|
206
|
+
continuationId: number;
|
|
207
|
+
matchedBytes: number;
|
|
208
|
+
contextTruncated: boolean;
|
|
209
|
+
continuationTruncated: boolean;
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
/** {@link CorpusResult} as text, plus the prose for why nothing matched. The
|
|
213
|
+
* byte layer reports a STATE; saying it in words belongs to the text layer. */
|
|
214
|
+
export interface CorpusTextResult {
|
|
215
|
+
query: string;
|
|
216
|
+
pairs: CorpusTextPair[];
|
|
217
|
+
resolved: number;
|
|
218
|
+
reached: number;
|
|
219
|
+
totalContexts: number;
|
|
220
|
+
browsed: boolean;
|
|
221
|
+
note?: string;
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
/** What the text helper says when the byte layer reports a miss. */
|
|
225
|
+
const CORPUS_NOTE: Record<string, string> = {
|
|
226
|
+
"nothing-resolved":
|
|
227
|
+
"No trained note sits above the parts of that text the mind recognised. " +
|
|
228
|
+
"It addresses content exactly, so try wording closer to something it was " +
|
|
229
|
+
"actually given — or browse the examples instead.",
|
|
230
|
+
"no-continuations":
|
|
231
|
+
"That text reaches stored nodes, but none of them carries a learnt " +
|
|
232
|
+
"continuation.",
|
|
233
|
+
};
|
|
234
|
+
|
|
235
|
+
/** UTF-8 of bytes for display: reuse {@link decodeText} (the mind's own text
|
|
236
|
+
* conversion), then drop the replacement character a byte-boundary cut leaves
|
|
237
|
+
* behind. Much of a real corpus is non-Latin, so that trailing U+FFFD is the
|
|
238
|
+
* common case, not an exotic one — and it is the ONLY thing added here. */
|
|
239
|
+
function previewCorpusText(bytes: Uint8Array): string {
|
|
240
|
+
return decodeText(bytes).replace(/\uFFFD+$/, "").replace(/\s+/g, " ").trim();
|
|
241
|
+
}
|
|
242
|
+
|
|
197
243
|
export interface MindOptions {
|
|
198
244
|
seed?: number;
|
|
199
245
|
recallQueryK?: number;
|
|
200
246
|
haloQueryK?: number;
|
|
247
|
+
/** Items one rationale step may itemise — see {@link MindConfig}. */
|
|
248
|
+
rationaleSampleK?: number;
|
|
249
|
+
/** Corpus-reading capacities and budgets — see {@link MindConfig}. */
|
|
250
|
+
corpusLimitMax?: number;
|
|
251
|
+
corpusClimbs?: number;
|
|
252
|
+
corpusContextsPerClimb?: number;
|
|
253
|
+
corpusSampleProbes?: number;
|
|
254
|
+
corpusPreviewBytes?: number;
|
|
255
|
+
corpusSampleFloorBytes?: number;
|
|
201
256
|
normalizeEpsilon?: number;
|
|
202
257
|
cosineEpsilon?: number;
|
|
203
258
|
geometry?: Partial<import("../config.js").GeometryConfig>;
|
|
@@ -211,13 +266,12 @@ export interface MindOptions {
|
|
|
211
266
|
host: import("../extension.js").ExtensionHost,
|
|
212
267
|
) => import("./pipeline-mechanism.js").PipelineMechanism)[];
|
|
213
268
|
/** Measure the computational usage of every inference call — see
|
|
214
|
-
* src/meter.ts.
|
|
215
|
-
*
|
|
216
|
-
*
|
|
217
|
-
*
|
|
218
|
-
*
|
|
219
|
-
*
|
|
220
|
-
* AGENTS §2.11), so profile without a trace. */
|
|
269
|
+
* src/meter.ts. Off by default and free when off (one null check per store
|
|
270
|
+
* read); on, each `respond`/`respondTurn` leaves a {@link Mind.lastCost}
|
|
271
|
+
* report behind. Counters are deterministic, so two runs of the same query on
|
|
272
|
+
* the same store are diffable; the millisecond fields are not. Profiling
|
|
273
|
+
* NEVER changes an answer — but attaching a RATIONALE does: a traced response
|
|
274
|
+
* bypasses the ctx memos (memoization.md), so profile without a trace. */
|
|
221
275
|
profile?: boolean;
|
|
222
276
|
/** Content canonicalizer applied to EVERY response (any modality) for
|
|
223
277
|
* equivalence-class resolution — see src/canon.ts. Text entry points
|
|
@@ -348,6 +402,20 @@ export class Mind implements MindContext {
|
|
|
348
402
|
* `traverse.ts`'s ONE definition (edge or halo, with its response-scoped
|
|
349
403
|
* cache). The search holds a bare Store and cannot reach that cache itself,
|
|
350
404
|
* so it asks through this hook; a bare host keeps its raw-store fallback. */
|
|
405
|
+
/** The canonical identity for the search (see GraphSearchHost). */
|
|
406
|
+
canonResolve(bytes: Uint8Array): number | null {
|
|
407
|
+
return canonResolveImpl(this, bytes);
|
|
408
|
+
}
|
|
409
|
+
|
|
410
|
+
/** Feed a search refusal into the rationale (see GraphSearchHost). */
|
|
411
|
+
reportSearch(
|
|
412
|
+
name: string,
|
|
413
|
+
parts: ReadonlyArray<Uint8Array>,
|
|
414
|
+
note: string,
|
|
415
|
+
): void {
|
|
416
|
+
this.trace?.step(name, parts.map((b) => rItem(b)), [], note);
|
|
417
|
+
}
|
|
418
|
+
|
|
351
419
|
leadsSomewhere(id: number): boolean {
|
|
352
420
|
return leadsSomewhere(this, id);
|
|
353
421
|
}
|
|
@@ -375,6 +443,11 @@ export class Mind implements MindContext {
|
|
|
375
443
|
* with the most distributional evidence (highest `prevOf` count — the
|
|
376
444
|
* structural manifestation of its halo). When evidence is equal the
|
|
377
445
|
* first-inserted edge wins. */
|
|
446
|
+
/** See {@link GraphSearchHost.contentCuts}. */
|
|
447
|
+
contentCuts(bytes: Uint8Array): readonly number[] {
|
|
448
|
+
return contentBoundaries(this.space, bytes);
|
|
449
|
+
}
|
|
450
|
+
|
|
378
451
|
chooseNext(node: number): number | undefined {
|
|
379
452
|
return chooseNext(this, node, this._edgeGuide);
|
|
380
453
|
}
|
|
@@ -738,6 +811,55 @@ export class Mind implements MindContext {
|
|
|
738
811
|
return decodeText(r.bytes);
|
|
739
812
|
}
|
|
740
813
|
|
|
814
|
+
// ── Reading the trained memory back ─────────────────────────────────────
|
|
815
|
+
|
|
816
|
+
/** Which stored notes does this query REACH? BYTES in, BYTES out — this
|
|
817
|
+
* method has no notion of text or encoding; the text case is
|
|
818
|
+
* {@link searchCorpusText}, which is one caller of this.
|
|
819
|
+
*
|
|
820
|
+
* Exact content addressing through the machinery an answer already uses
|
|
821
|
+
* (see src/mind/corpus.ts): the query's recognised sites are the resolved
|
|
822
|
+
* subtrees, the climb goes up from the biggest, and a result is a context
|
|
823
|
+
* that carries a learnt continuation. Nothing is written and nothing is
|
|
824
|
+
* indexed. */
|
|
825
|
+
searchCorpus(queryBytes: Uint8Array, limit?: number): CorpusResult {
|
|
826
|
+
return searchCorpus(this, queryBytes, limit);
|
|
827
|
+
}
|
|
828
|
+
|
|
829
|
+
/** Browse real pairs. Deterministic: `from` is the caller's own offset in
|
|
830
|
+
* [0,1), so browsing twice with different offsets shows different notes
|
|
831
|
+
* without a random draw. */
|
|
832
|
+
sampleCorpus(limit?: number, from?: number): CorpusResult {
|
|
833
|
+
return sampleCorpus(this, limit, from);
|
|
834
|
+
}
|
|
835
|
+
|
|
836
|
+
/** The TEXT case of {@link searchCorpus}: encode, search, decode. The search
|
|
837
|
+
* itself exists once, in the byte layer above; only the rendering lives
|
|
838
|
+
* here, with the rest of this class's text modality. */
|
|
839
|
+
searchCorpusText(query: string, limit?: number): CorpusTextResult {
|
|
840
|
+
const result = this.searchCorpus(
|
|
841
|
+
new TextEncoder().encode(query),
|
|
842
|
+
limit,
|
|
843
|
+
);
|
|
844
|
+
return {
|
|
845
|
+
query,
|
|
846
|
+
pairs: result.pairs.map((p: CorpusPair): CorpusTextPair => ({
|
|
847
|
+
context: previewCorpusText(p.context),
|
|
848
|
+
continuation: previewCorpusText(p.continuation),
|
|
849
|
+
contextId: p.contextId,
|
|
850
|
+
continuationId: p.continuationId,
|
|
851
|
+
matchedBytes: p.matchedBytes,
|
|
852
|
+
contextTruncated: p.contextTruncated,
|
|
853
|
+
continuationTruncated: p.continuationTruncated,
|
|
854
|
+
})),
|
|
855
|
+
resolved: result.resolved,
|
|
856
|
+
reached: result.reached,
|
|
857
|
+
totalContexts: result.totalContexts,
|
|
858
|
+
browsed: result.browsed,
|
|
859
|
+
note: result.miss === "matched" ? undefined : CORPUS_NOTE[result.miss],
|
|
860
|
+
};
|
|
861
|
+
}
|
|
862
|
+
|
|
741
863
|
// ── Conversation API ────────────────────────────────────────────────────
|
|
742
864
|
|
|
743
865
|
/** Begin a new conversation, optionally restoring from a previously-saved
|
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
// a list of PipelineMechanism objects — it never imports a mechanism-specific
|
|
6
6
|
// type and never has a special-case branch for any mechanism.
|
|
7
7
|
//
|
|
8
|
-
// The four constraints of the free-will architecture (
|
|
8
|
+
// The four constraints of the free-will architecture (mechanism-market.md):
|
|
9
9
|
// 1. DECOUPLING — mechanisms import nothing from each other or from pipeline.
|
|
10
10
|
// 2. DECLARED COMPETENCE — floor() returns null when impossible, a number when
|
|
11
11
|
// possible. Binary, auditable, no learned scores.
|
|
@@ -148,17 +148,17 @@ export class Precomputed {
|
|
|
148
148
|
);
|
|
149
149
|
}
|
|
150
150
|
|
|
151
|
-
// REMOVED — the WIDE exhaustive-√N resonance list (`wideResonance`).
|
|
151
|
+
// REMOVED — the WIDE exhaustive-√N resonance list (`wideResonance`). It ran
|
|
152
152
|
// `resonate(guide, √N, exhaustive=true)` whenever the top hit cleared
|
|
153
|
-
// conceptThreshold, so consumers could look "past the top-k".
|
|
153
|
+
// conceptThreshold, so consumers could look "past the top-k". Every consumer
|
|
154
154
|
// only ever needed ≤ 2·recallQueryK proposals (the substitution bridge's own
|
|
155
155
|
// candidate cap) or a content-addressed answer (prefix completion's
|
|
156
|
-
// formsOpenedBy), and every proposal is byte-verified downstream
|
|
157
|
-
// the exhaustive scan bought recall at O(index)
|
|
158
|
-
// measured: 244K annVectorReads per refusing query,
|
|
159
|
-
// byte-identical to a top-k read.
|
|
160
|
-
// (the one top-k read) and the write side's window index
|
|
161
|
-
// recall.ts and prefix-completion.ts.
|
|
156
|
+
// formsOpenedBy), and every proposal is byte-verified downstream
|
|
157
|
+
// (exact-vs-approximate.md), so the exhaustive scan bought recall at O(index)
|
|
158
|
+
// cost for an O(k) need — measured: 244K annVectorReads per refusing query,
|
|
159
|
+
// ~1.5 s, every answer byte-identical to a top-k read. The two consumers now
|
|
160
|
+
// read `resonance()` (the one top-k read) and the write side's window index
|
|
161
|
+
// respectively — see recall.ts and prefix-completion.ts.
|
|
162
162
|
|
|
163
163
|
private _frames?: Promise<ReadonlyArray<FrameInstance>>;
|
|
164
164
|
/** THE FRAME INVENTORY — every ranked candidate that reads as an instance of
|
|
@@ -166,14 +166,16 @@ export class Precomputed {
|
|
|
166
166
|
* ({@link FrameInstance}). The one place the engine represents "a position
|
|
167
167
|
* whose occupant comes from the context rather than the corpus".
|
|
168
168
|
*
|
|
169
|
-
*
|
|
170
|
-
*
|
|
171
|
-
*
|
|
172
|
-
*
|
|
173
|
-
*
|
|
174
|
-
*
|
|
175
|
-
*
|
|
176
|
-
*
|
|
169
|
+
* AN INVENTORY, NOT AN ELECTION. It reports every pairing and elects no
|
|
170
|
+
* frame,
|
|
171
|
+
* deliberately: a slot is a property of a PAIRING, not of the query, and
|
|
172
|
+
* different candidates put slots in different places. Committing to one
|
|
173
|
+
* reading here would push whichever consumer asked first onto everyone else —
|
|
174
|
+
* the market's decoupling (mechanism-market.md) broken from inside the shared
|
|
175
|
+
* container, and the population error commonality.md names. Each consumer
|
|
176
|
+
* groups and commits for its own question; reference elects the modal slot
|
|
177
|
+
* signature, and a consumer wanting a different reading is not fighting this
|
|
178
|
+
* one.
|
|
177
179
|
*
|
|
178
180
|
* NO LICENCE EITHER. Knowing a span is variable is safe for every consumer
|
|
179
181
|
* — it can only improve an alignment. Knowing one may be VOICED through is
|
|
@@ -189,9 +191,10 @@ export class Precomputed {
|
|
|
189
191
|
const capBytes = this.query.length * W;
|
|
190
192
|
const out: FrameInstance[] = [];
|
|
191
193
|
for (const h of await this.resonance()) {
|
|
192
|
-
// REJECT BY LENGTH BEFORE RECONSTRUCTING (
|
|
193
|
-
// indexed read, `bytesPrefix` rebuilds a subtree.
|
|
194
|
-
// cap is applied — it is a bounded-read
|
|
194
|
+
// REJECT BY LENGTH BEFORE RECONSTRUCTING (bounded-reads.md):
|
|
195
|
+
// `contentLen` is an indexed read, `bytesPrefix` rebuilds a subtree.
|
|
196
|
+
// ONLY the phrase-scale cap is applied — it is a bounded-read
|
|
197
|
+
// discipline, not a judgement.
|
|
195
198
|
//
|
|
196
199
|
// A LOWER bound was here too (`dominates(len, query.length)`, on the
|
|
197
200
|
// reasoning that a candidate shorter than half the query cannot supply
|
|
@@ -519,7 +522,8 @@ function computeWeave(
|
|
|
519
522
|
// IDF — gates the aligner has no equivalent of.
|
|
520
523
|
//
|
|
521
524
|
// So the climb PROPOSES the pairing (which structure, which query span) and
|
|
522
|
-
// bytes DECIDE its terms (
|
|
525
|
+
// bytes DECIDE its terms (exact-vs-approximate.md). Three gates, each one
|
|
526
|
+
// measured:
|
|
523
527
|
//
|
|
524
528
|
// • it may only take query bytes NO literal run claimed. Run inline with
|
|
525
529
|
// phase 1 this did the opposite of "exact decides" — a higher-ranked
|
package/src/mind/pipeline.ts
CHANGED
|
@@ -130,15 +130,15 @@ export interface NarrowDecisionData {
|
|
|
130
130
|
}
|
|
131
131
|
|
|
132
132
|
/** Structured payload of the "regimePrediction" rationale step — the R8
|
|
133
|
-
* observation exposed as data.
|
|
134
|
-
*
|
|
135
|
-
*
|
|
136
|
-
*
|
|
137
|
-
*
|
|
138
|
-
*
|
|
139
|
-
*
|
|
140
|
-
*
|
|
141
|
-
*
|
|
133
|
+
* observation exposed as data. After the first mechanism (cover, which
|
|
134
|
+
* mechanism-market.md runs first) grounds or abstains, the market's whole
|
|
135
|
+
* outcome is already determined by the one cost ladder: the consensus climb
|
|
136
|
+
* runs exactly when `worthRunning(2 * STEP)` is true — CAST (floor 2·STEP) is
|
|
137
|
+
* the cheapest mechanism that first-touches it, and confluence (3·STEP) /
|
|
138
|
+
* extraction (CONCEPT+STEP) are only reached after CAST is. An incumbent at or
|
|
139
|
+
* below that floor prunes CAST and, with it, the climb (retrieval); anything
|
|
140
|
+
* above — or no incumbent — runs the full market and the climb (composition).
|
|
141
|
+
* Purely observational; never read by inference. */
|
|
142
142
|
export interface RegimePredictionData {
|
|
143
143
|
version: 1;
|
|
144
144
|
/** retrieval | composition — the two regimes R1 measured as a ~100× cost
|
|
@@ -187,13 +187,13 @@ export async function think(
|
|
|
187
187
|
// ── Pre-computation ──────────────────────────────────────────────────
|
|
188
188
|
const mechanisms = mechs ?? defaultMechanisms;
|
|
189
189
|
const meter = ctx.meter;
|
|
190
|
-
// recognition is a shared analysis (
|
|
190
|
+
// recognition is a shared analysis (meter.md contract 5): it does the query's
|
|
191
191
|
// own store work (perceive → foldTree → resolve), which used to land in
|
|
192
192
|
// `think` and in nothing narrower — the meter's one accounting surface must
|
|
193
193
|
// charge it to itself, exactly as attention/weave/resonance are charged.
|
|
194
|
-
// SYNCHRONOUS phase: recognition is on the sync side of
|
|
195
|
-
// is timed with `timeSync` — wrapping it in a promise would make a
|
|
196
|
-
// response await where an unprofiled one does not.
|
|
194
|
+
// SYNCHRONOUS phase: recognition is on the sync side of meter.md's seam, so
|
|
195
|
+
// it is timed with `timeSync` — wrapping it in a promise would make a
|
|
196
|
+
// profiled response await where an unprofiled one does not.
|
|
197
197
|
const rec = meter
|
|
198
198
|
? meter.timeSync("recognise", () => recognise(ctx, query))
|
|
199
199
|
: recognise(ctx, query);
|
|
@@ -225,15 +225,15 @@ export async function think(
|
|
|
225
225
|
}
|
|
226
226
|
}
|
|
227
227
|
|
|
228
|
-
// Phase 2: the shared pre-computation container.
|
|
229
|
-
// (recognition, computed spans, guide) — every expensive analysis
|
|
230
|
-
//
|
|
231
|
-
//
|
|
232
|
-
//
|
|
233
|
-
//
|
|
234
|
-
//
|
|
235
|
-
//
|
|
236
|
-
//
|
|
228
|
+
// Phase 2: the shared pre-computation container. Eager fields only
|
|
229
|
+
// (recognition, computed spans, guide) — every expensive analysis (consensus
|
|
230
|
+
// climb, weave, span-shape classification) is a lazily-cached method on
|
|
231
|
+
// Precomputed, first-touched by whichever mechanism's floor survives its
|
|
232
|
+
// cheap gates and the worthRunning check. A query no mechanism climbs for
|
|
233
|
+
// (e.g. one an extension decided) never climbs. NOT phased: the constructor
|
|
234
|
+
// itself is trivial (it only derives `k`), so a phase here would add a
|
|
235
|
+
// zero-work entry to every profiled report — the meter attributes WORK
|
|
236
|
+
// (meter.md); the trace already represents structure.
|
|
237
237
|
const pre = new Precomputed(ctx, query, rec, computed, ctx._edgeGuide);
|
|
238
238
|
|
|
239
239
|
// ── Grounding: ONE lightest-derivation choice among the mechanisms ────
|
|
@@ -298,16 +298,17 @@ export async function think(
|
|
|
298
298
|
const worthRunning = (floor: number) =>
|
|
299
299
|
best === null || grade(floor) < grade(best.weight);
|
|
300
300
|
|
|
301
|
-
// REGIME PREDICTION (R8) — observational only.
|
|
302
|
-
// had its turn (cover, which
|
|
303
|
-
// outcome is already determined by the one cost ladder: the
|
|
304
|
-
// runs exactly when `worthRunning(2 * STEP)` is true — CAST
|
|
305
|
-
// the cheapest mechanism that first-touches it, so an
|
|
306
|
-
// grade 2 prunes CAST and, with it, confluence (3·STEP)
|
|
307
|
-
// (CONCEPT+STEP) (retrieval); anything above — or no incumbent
|
|
308
|
-
// full market and the climb (composition).
|
|
309
|
-
// the same function the loop itself uses — nothing is
|
|
310
|
-
// engine had not already computed, and nothing is read
|
|
301
|
+
// REGIME PREDICTION (R8) — observational only. Once the FIRST mechanism has
|
|
302
|
+
// had its turn (cover, which mechanism-market.md places first and floors at
|
|
303
|
+
// 0), the market's outcome is already determined by the one cost ladder: the
|
|
304
|
+
// consensus climb runs exactly when `worthRunning(2 * STEP)` is true — CAST
|
|
305
|
+
// (floor 2·STEP) is the cheapest mechanism that first-touches it, so an
|
|
306
|
+
// incumbent at or below grade 2 prunes CAST and, with it, confluence (3·STEP)
|
|
307
|
+
// and extraction (CONCEPT+STEP) (retrieval); anything above — or no incumbent
|
|
308
|
+
// — runs the full market and the climb (composition). The predicate is
|
|
309
|
+
// `worthRunning`, the same function the loop itself uses — nothing is
|
|
310
|
+
// computed here that the engine had not already computed, and nothing is read
|
|
311
|
+
// back by inference.
|
|
311
312
|
//
|
|
312
313
|
// EMITTED BEFORE THE SECOND MECHANISM'S FLOOR, never after some mechanism's
|
|
313
314
|
// run: a "prediction" published after the fact could assert "the climb will
|
|
@@ -544,12 +545,40 @@ export async function think(
|
|
|
544
545
|
ctx.store.nextFirst(id, hubBound(ctx)).map((n) => read(ctx, n))
|
|
545
546
|
)
|
|
546
547
|
: [];
|
|
548
|
+
// REPORTABLE, NOT SILENT. A declared-complete grounding ends the derivation
|
|
549
|
+
// here, and that decision is part of the derivation's shape: the reader of a
|
|
550
|
+
// rationale must be able to see that the chain stopped because the mechanism
|
|
551
|
+
// claimed the query WAS the context, not because nothing followed. The step
|
|
552
|
+
// carries the claim, not a re-description of the answer — the extension is
|
|
553
|
+
// skipped, so there is no output item to show.
|
|
554
|
+
if (decided.complete) {
|
|
555
|
+
ctx.trace?.step(
|
|
556
|
+
"completeGrounding",
|
|
557
|
+
[rItem(answer, provenance)],
|
|
558
|
+
[],
|
|
559
|
+
"grounding declared complete — the query IS the context, so " +
|
|
560
|
+
"post-grounding extension is skipped",
|
|
561
|
+
);
|
|
562
|
+
}
|
|
563
|
+
// THE REASONER JUDGES ITS OWN EXTENSIONS BY THE PIPELINE'S REMAINDER, not by
|
|
564
|
+
// the ladder's `accounted` — and by the SAME reading the fuse gate below uses,
|
|
565
|
+
// with the same W floor. `accounted` is a COST quantity (measured: a query
|
|
566
|
+
// fully explained by one computed span plus bridged connectors reports
|
|
567
|
+
// `accounted: []` while nothing is unexplained), and a remainder under one
|
|
568
|
+
// river-fold quantum is bridging punctuation, never a second topic — so it
|
|
569
|
+
// licenses no extension and blocks none.
|
|
570
|
+
const explained: Array<[number, number]> = [
|
|
571
|
+
...decided.accounted,
|
|
572
|
+
...pre.computed.map((u): [number, number] => [u.i, u.j]),
|
|
573
|
+
];
|
|
574
|
+
const uncovered = unexplainedSpans(query.length, explained)
|
|
575
|
+
.filter(([a, b]) => b - a >= ctx.space.maxGroup);
|
|
547
576
|
const reasoned = decided.complete ? answer : meter
|
|
548
577
|
? await meter.time(
|
|
549
578
|
"reason",
|
|
550
|
-
() => reason(ctx, query, answer, preConsumed, pre, voiced),
|
|
579
|
+
() => reason(ctx, query, answer, preConsumed, pre, voiced, uncovered),
|
|
551
580
|
)
|
|
552
|
-
: await reason(ctx, query, answer, preConsumed, pre, voiced);
|
|
581
|
+
: await reason(ctx, query, answer, preConsumed, pre, voiced, uncovered);
|
|
553
582
|
|
|
554
583
|
// Fuse only when the query has a genuine REMAINDER no mechanism's
|
|
555
584
|
// structural evidence touched at all. `decided.accounted` alone
|
|
@@ -567,10 +596,6 @@ export async function think(
|
|
|
567
596
|
// observed: a single space between two fully-computed arithmetic spans
|
|
568
597
|
// ("2+2 3+3") registered as "unaccounted" and pulled in an unrelated
|
|
569
598
|
// corpus fact, corrupting "4 6" into "4 63".
|
|
570
|
-
const explained: Array<[number, number]> = [
|
|
571
|
-
...decided.accounted,
|
|
572
|
-
...pre.computed.map((u): [number, number] => [u.i, u.j]),
|
|
573
|
-
];
|
|
574
599
|
const remainder = unaccounted(explained);
|
|
575
600
|
// Whether the winning candidate's entire recognised substance is
|
|
576
601
|
// COMPUTED — every accounted span exactly a pre.computed span, nothing
|
package/src/mind/primitives.ts
CHANGED
|
@@ -59,11 +59,11 @@ export function perceiveKey(
|
|
|
59
59
|
/** Perceive input into a content-defined tree (the river fold).
|
|
60
60
|
* Deterministic — identical bytes always produce an identical tree.
|
|
61
61
|
*
|
|
62
|
-
*
|
|
63
|
-
*
|
|
64
|
-
*
|
|
65
|
-
*
|
|
66
|
-
*
|
|
62
|
+
* `boundaries` is an optional sorted list of proper byte offsets where the fold
|
|
63
|
+
* must split so that each prefix segment folds identically to how it folded
|
|
64
|
+
* when it was learned (fold-contract.md stable-prefix contract). Only the
|
|
65
|
+
* CALLER — who assembled the multi-turn context — knows where those boundaries
|
|
66
|
+
* are; the geometry never guesses them from the bytes. */
|
|
67
67
|
export function perceive(
|
|
68
68
|
ctx: MindContext,
|
|
69
69
|
input: Input,
|
package/src/mind/reasoning.ts
CHANGED
|
@@ -43,6 +43,10 @@ export async function reason(
|
|
|
43
43
|
preConsumed: ReadonlySet<number>,
|
|
44
44
|
pre: Precomputed,
|
|
45
45
|
voiced: readonly Uint8Array[] = [],
|
|
46
|
+
/** The query material the GROUNDING left uncovered — the cost ladder's own
|
|
47
|
+
* `unaccounted` spans. Only the reasoner's OWN extensions are judged
|
|
48
|
+
* against it; a mechanism carrying its own `used` set owns its shape. */
|
|
49
|
+
uncovered: readonly (readonly [number, number])[] = [],
|
|
46
50
|
): Promise<Uint8Array> {
|
|
47
51
|
// Echo guard: a query that is ITSELF a learnt continuation (some context's
|
|
48
52
|
// answer) is being asked back at the system — hopping forward from it would
|
|
@@ -200,6 +204,57 @@ export async function reason(
|
|
|
200
204
|
const fc = await follow(ctx, pivot, qv);
|
|
201
205
|
consumeAll(pivot);
|
|
202
206
|
if (fc === null || bytesEqual(fc, cur) || restatesQuery(query, fc)) break;
|
|
207
|
+
// WHOSE EXTENSION IS THIS?
|
|
208
|
+
//
|
|
209
|
+
// `voiced` is what the mechanism WITHHELD (the pipeline sends the used
|
|
210
|
+
// anchors' CONTINUATIONS, not their bytes — see pipeline's own note), so a
|
|
211
|
+
// non-empty `voiced` means exactly what that note says: the grounding came
|
|
212
|
+
// from a mechanism that carries its own short `used` set (cast/join) and
|
|
213
|
+
// therefore owns the shape of its answer. The further terms inside such a
|
|
214
|
+
// seat are legitimately followable — test/29 C3's `Mona Lisa` lives inside
|
|
215
|
+
// the voiced seat and leads on to a fact about neither analog.
|
|
216
|
+
//
|
|
217
|
+
// Every other grounding is ordinary, and an extension of it is the
|
|
218
|
+
// reasoner's own inference: it is taken only while question material the
|
|
219
|
+
// grounding left uncovered remains AND the step carries some of it, judged
|
|
220
|
+
// by the mind's own line between chance and evidence — one W-byte window,
|
|
221
|
+
// no word notion, no character class, no threshold. Measured: the drift's
|
|
222
|
+
// second step (`the Eiffel Tower is in Paris` after `Paris is famous for
|
|
223
|
+
// the Eiffel Tower`) carries no window of `" famous for"` and is refused,
|
|
224
|
+
// while the first carries it. Terminates by a real argument: the uncovered
|
|
225
|
+
// material is finite and each taken extension must carry some of it.
|
|
226
|
+
const producerOwnsShape = voiced.length > 0;
|
|
227
|
+
if (!producerOwnsShape && uncovered.length > 0) {
|
|
228
|
+
const W = ctx.space.maxGroup;
|
|
229
|
+
let progress = false;
|
|
230
|
+
for (const [a, b] of uncovered) {
|
|
231
|
+
for (let i = a; i + W <= b && !progress; i++) {
|
|
232
|
+
if (indexOf(fc, query.subarray(i, i + W), 0) >= 0) progress = true;
|
|
233
|
+
}
|
|
234
|
+
if (progress) break;
|
|
235
|
+
}
|
|
236
|
+
if (!progress) {
|
|
237
|
+
// THE BRAKE, MADE VISIBLE. The reasoner declines a step that carries
|
|
238
|
+
// none of the material the grounding left uncovered — the drift the
|
|
239
|
+
// extension tests pin. A refusal that leaves no trace is the kind of
|
|
240
|
+
// silent cut AGENTS §6 forbids: the rationale is where a reader learns
|
|
241
|
+
// that an extension was declined for want of question material, and
|
|
242
|
+
// where the next person sees why the chain stopped here. Measured with
|
|
243
|
+
// the check disabled, test/110 and test/116 fail — so this brake is the
|
|
244
|
+
// only thing keeping the extension honest until the pivot reports its
|
|
245
|
+
// own accounted spans and the ladder can judge it instead.
|
|
246
|
+
const left = uncovered.reduce((n, [a, b]) => n + (b - a), 0);
|
|
247
|
+
ctx.trace?.step(
|
|
248
|
+
"pivotRefused",
|
|
249
|
+
[rItem(cur, "answer"), rItem(query, "query")],
|
|
250
|
+
uncovered.map(([a, b]) => rItem(query.subarray(a, b), "uncovered")),
|
|
251
|
+
`the step carries none of the question material the grounding left ` +
|
|
252
|
+
`uncovered (${left} byte(s) in ${uncovered.length} span(s)) — refused`,
|
|
253
|
+
);
|
|
254
|
+
break;
|
|
255
|
+
}
|
|
256
|
+
}
|
|
257
|
+
if (ctx.meter) ctx.meter.pivotSteps++;
|
|
203
258
|
t ??= ctx.trace?.enter("reason", [rItem(startedFrom, "grounded")]);
|
|
204
259
|
ctx.trace?.step(
|
|
205
260
|
"pivotStep",
|
package/src/mind/recognition.ts
CHANGED
|
@@ -34,19 +34,20 @@ import type { Leaf, Site } from "./graph-search.js";
|
|
|
34
34
|
*
|
|
35
35
|
* Both O(n · maxGroup) bounded O(1) probes — never a scan of the corpus.
|
|
36
36
|
*
|
|
37
|
-
*
|
|
38
|
-
*
|
|
39
|
-
*
|
|
40
|
-
*
|
|
41
|
-
*
|
|
42
|
-
*
|
|
43
|
-
*
|
|
44
|
-
*
|
|
45
|
-
*
|
|
46
|
-
*
|
|
47
|
-
*
|
|
48
|
-
*
|
|
49
|
-
*
|
|
37
|
+
* ONE READING PER BYTE STREAM, deliberately: there is no "cheap mode" that
|
|
38
|
+
* skips
|
|
39
|
+
* the edge-trim fallbacks. A `trimmed` variant was tried and REFUTED twice
|
|
40
|
+
* over. Its premise — "the trims only recover misaligned FRAGMENTS, so a
|
|
41
|
+
* consumer whose gate rejects fragments loses nothing" — is false: the
|
|
42
|
+
* left/right trim loops below exist precisely to find WHOLE trained forms
|
|
43
|
+
* embedded at an offset the query's own fold did not cut, and such a form has
|
|
44
|
+
* no structural parents or containers, so it passes the pivot's fragment gate
|
|
45
|
+
* and is exactly the candidate a multi-hop chain steps through. Skipping them
|
|
46
|
+
* narrows the pivot's evidence silently. And a per-caller variant has to key
|
|
47
|
+
* the memo by the variant, which breaks the "computed at most once" property
|
|
48
|
+
* (memoization.md): the pipeline recognises a grounded answer untrimmed for
|
|
49
|
+
* `preConsumed`, and the pivot then recognises the same bytes again — the
|
|
50
|
+
* saving inverts into a doubling on the path it was measured for. */
|
|
50
51
|
export function recognise(
|
|
51
52
|
ctx: MindContext,
|
|
52
53
|
bytes: Uint8Array,
|
|
@@ -180,16 +181,15 @@ function recogniseImpl(ctx: MindContext, bytes: Uint8Array): Recognition {
|
|
|
180
181
|
// and read the answer from `starts`, which is exactly {0, W, 2W, …}
|
|
181
182
|
// because riverFold groups fixed-arity — arithmetic, not evidence.
|
|
182
183
|
//
|
|
183
|
-
// Measured on the 17.9M-node store, over the sites of 7 probes (1 good,
|
|
184
|
-
//
|
|
185
|
-
//
|
|
186
|
-
//
|
|
187
|
-
//
|
|
188
|
-
//
|
|
189
|
-
//
|
|
190
|
-
//
|
|
191
|
-
//
|
|
192
|
-
// rarity does not separate: "hi" has 1 container, "the" 572
|
|
184
|
+
// Measured on the 17.9M-node store, over the sites of 7 probes (1 good, 11
|
|
185
|
+
// junk by hand-labelling, corrected for whole-query forms): len >= W
|
|
186
|
+
// rejects "hi"(2) "of"(2) "is"(2) "di"(2) "the"(3), admits "Eiffel
|
|
187
|
+
// Tower"(12) and both whole-query forms len >= W-1 admits "the" — W-1 is
|
|
188
|
+
// the write side's straddle neighbour for RETRIEVAL, never a claim about
|
|
189
|
+
// units commonality.md saturation admits 11/11 junk: edgeAncestors on a
|
|
190
|
+
// site node reaches 1..48 contexts, so dominates(ctx, N) needs ctx > 162805
|
|
191
|
+
// and never fires; every site reads DISC rarity does not separate: "hi" has
|
|
192
|
+
// 1 container, "the" 572
|
|
193
193
|
//
|
|
194
194
|
// A span covering the WHOLE query is exempt: then it is not a fragment of
|
|
195
195
|
// something longer, it is the question ("hi" asked on its own).
|