@hviana/sema 0.7.9 → 0.8.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +22 -1
- package/DATASETS.md +1 -1
- package/dist/example/train_base/config.js +2 -2
- package/dist/example/train_base/corpora/massive.js +1 -1
- package/dist/example/train_base/readers.js +1 -1
- package/dist/src/geometry.d.ts +10 -10
- package/dist/src/geometry.js +25 -24
- package/dist/src/meter.d.ts +4 -12
- package/dist/src/meter.js +14 -14
- package/dist/src/mind/attention.js +12 -12
- package/dist/src/mind/bridge.d.ts +8 -8
- package/dist/src/mind/bridge.js +33 -32
- package/dist/src/mind/graph-search.d.ts +0 -8
- package/dist/src/mind/graph-search.js +38 -25
- package/dist/src/mind/junction.d.ts +1 -1
- package/dist/src/mind/junction.js +8 -8
- package/dist/src/mind/learning.js +36 -35
- package/dist/src/mind/match.js +14 -13
- package/dist/src/mind/mechanisms/cover.js +13 -12
- package/dist/src/mind/mechanisms/prefix-completion.js +24 -24
- package/dist/src/mind/mechanisms/recall.js +38 -40
- package/dist/src/mind/mechanisms/reference.js +16 -16
- package/dist/src/mind/mind.d.ts +6 -7
- package/dist/src/mind/pipeline-mechanism.d.ts +10 -8
- package/dist/src/mind/pipeline-mechanism.js +25 -21
- package/dist/src/mind/pipeline.d.ts +9 -9
- package/dist/src/mind/pipeline.js +24 -23
- package/dist/src/mind/primitives.d.ts +5 -5
- package/dist/src/mind/primitives.js +5 -5
- package/dist/src/mind/recognition.d.ts +14 -13
- package/dist/src/mind/recognition.js +53 -38
- package/dist/src/mind/resonance.js +21 -21
- package/dist/src/mind/traverse.d.ts +54 -52
- package/dist/src/mind/traverse.js +74 -72
- package/dist/src/mind/types.d.ts +4 -4
- package/dist/src/store.d.ts +12 -12
- package/dist/src/store.js +12 -12
- package/docs/INDEX.md +2 -2
- package/docs/architecture/exact-vs-approximate.md +2 -1
- package/docs/architecture/fold-contract.md +1 -1
- package/docs/failures/tempting-but-wrong.md +2 -3
- package/docs/harness/gates.md +7 -7
- package/example/train_base/config.ts +2 -2
- package/example/train_base/corpora/massive.ts +1 -1
- package/example/train_base/readers.ts +1 -1
- package/jsr.json +1 -1
- package/package.json +1 -1
- package/src/geometry.ts +25 -24
- package/src/meter.ts +14 -14
- package/src/mind/attention.ts +12 -12
- package/src/mind/bridge.ts +33 -32
- package/src/mind/graph-search.ts +43 -24
- package/src/mind/junction.ts +8 -8
- package/src/mind/learning.ts +36 -35
- package/src/mind/match.ts +20 -19
- package/src/mind/mechanisms/cover.ts +13 -12
- package/src/mind/mechanisms/prefix-completion.ts +24 -24
- package/src/mind/mechanisms/recall.ts +38 -40
- package/src/mind/mechanisms/reference.ts +16 -16
- package/src/mind/mind.ts +6 -7
- package/src/mind/pipeline-mechanism.ts +25 -21
- package/src/mind/pipeline.ts +33 -32
- package/src/mind/primitives.ts +5 -5
- package/src/mind/recognition.ts +51 -36
- package/src/mind/resonance.ts +21 -21
- package/src/mind/traverse.ts +74 -72
- package/src/mind/types.ts +4 -4
- package/src/store.ts +20 -20
- package/test/08-storage.test.mjs +1 -1
- package/test/35-prefix-edge.test.mjs +1 -1
- package/test/40-choosenext-scale-guard.test.mjs +16 -17
- package/test/46-recognise-multibyte-edge.test.mjs +33 -0
- package/test/56-bridge-identity-admission.test.mjs +6 -6
- package/test/70-prefix-completion.test.mjs +4 -3
- package/test/72-prefix-candidate-supply.test.mjs +3 -3
- package/test/73-scaffolding-only-bridge-abstains.test.mjs +6 -6
- package/test/75-multiturn-context-optimisation.test.mjs +5 -5
- package/test/84-composed-answer-honesty.test.mjs +5 -6
- package/test/88-dependency-footprint.test.mjs +1 -1
- package/test/89-completion-recursion.test.mjs +17 -14
- package/test/90-connector-read-cap.test.mjs +10 -8
- package/test/93-regime-prediction.test.mjs +10 -10
- package/test/94-cross-region-budget.test.mjs +2 -2
- package/test/95-wide-resonance-removed.test.mjs +8 -7
- package/test/96-bytes-walk-termination.test.mjs +3 -3
- package/test/99-fact-join.test.mjs +38 -0
|
@@ -236,24 +236,22 @@ export async function recallByResonance(
|
|
|
236
236
|
}
|
|
237
237
|
}
|
|
238
238
|
|
|
239
|
-
// The query-relative grounding fraction, shared by tiers 2–4 — gated on
|
|
240
|
-
//
|
|
241
|
-
//
|
|
242
|
-
//
|
|
243
|
-
//
|
|
244
|
-
//
|
|
245
|
-
//
|
|
246
|
-
//
|
|
247
|
-
//
|
|
248
|
-
//
|
|
249
|
-
//
|
|
250
|
-
// √(lenG/lenQ)
|
|
251
|
-
//
|
|
252
|
-
//
|
|
253
|
-
//
|
|
254
|
-
//
|
|
255
|
-
// subtract the significance bar (3/√D, §8.3) before converting. Derived
|
|
256
|
-
// from the existing bars; never tuned.
|
|
239
|
+
// The query-relative grounding fraction, shared by tiers 2–4 — gated on the
|
|
240
|
+
// FRACTION OF THE QUERY the grounding explains, not the raw cosine. Root
|
|
241
|
+
// gists are unit vectors, but their magnitudes are recoverable from the byte
|
|
242
|
+
// lengths (‖·‖ = √len under the linear fold): cos = shared/√(lenQ·lenG), so
|
|
243
|
+
// shared/lenQ = cos·√(lenG/lenQ). The raw cosine punished honest containment
|
|
244
|
+
// — a query fully inside a longer grounded answer scored √(lenQ/lenG) and was
|
|
245
|
+
// refused — and let a long answer sharing only scaffolding pass; the
|
|
246
|
+
// query-relative fraction measures exactly what the reach bar means: how much
|
|
247
|
+
// of THE QUERY the store accounts for. Chance similarity survives the length
|
|
248
|
+
// conversion AMPLIFIED: the same √(lenG/lenQ) factor that converts an honest
|
|
249
|
+
// shared fraction into a query-relative one multiplies the estimator/chance
|
|
250
|
+
// floor too, so a long stored form (√(lenG/lenQ) ≈ 10 at 100×) lifted a
|
|
251
|
+
// noise-level cosine past the reach bar and grounded pure gibberish
|
|
252
|
+
// (observed). Only the ABOVE-CHANCE part of the similarity is evidence of
|
|
253
|
+
// shared content — subtract the significance bar (3/√D, thresholds.md) before
|
|
254
|
+
// converting. Derived from the existing bars; never tuned.
|
|
257
255
|
const sig = significanceBar(ctx.store.D);
|
|
258
256
|
const reach = reachThreshold(ctx.space.maxGroup);
|
|
259
257
|
const fracOfQuery = (cos: number, otherLen: number): number =>
|
|
@@ -411,14 +409,14 @@ export async function recallByResonance(
|
|
|
411
409
|
}
|
|
412
410
|
}
|
|
413
411
|
}
|
|
414
|
-
// 3b. Corroborated-substitution bridge — refusal-path only (bridge.ts).
|
|
415
|
-
//
|
|
416
|
-
//
|
|
417
|
-
//
|
|
418
|
-
//
|
|
419
|
-
//
|
|
420
|
-
// cluster here once made every honest refusal cost hundreds of ms
|
|
421
|
-
// of k.
|
|
412
|
+
// 3b. Corroborated-substitution bridge — refusal-path only (bridge.ts). The
|
|
413
|
+
// bridge's proposal source is the response's ONE top-k read — the same list
|
|
414
|
+
// recall already ranked above — never an exhaustive √N scan. The bridge's own
|
|
415
|
+
// candidate cap is 2·recallQueryK, so top-k proposals are exactly the budget
|
|
416
|
+
// it can consume, and every proposal is byte-verified downstream
|
|
417
|
+
// (exact-vs-approximate.md). Reuse the memoised `resonance()`; scanning every
|
|
418
|
+
// IVF cluster here once made every honest refusal cost hundreds of ms
|
|
419
|
+
// regardless of k.
|
|
422
420
|
const wideIds = async () => (await pre.resonance()).map((h) => h.id);
|
|
423
421
|
|
|
424
422
|
// Every gist-based tier has failed; before refusing, align the query
|
|
@@ -471,12 +469,12 @@ export async function recallByResonance(
|
|
|
471
469
|
// prefixCompletion runs a few lines below and carries the three guards
|
|
472
470
|
// this tier lacks — unreadable-continuation veto, sub-quantum
|
|
473
471
|
// continuation, and UNIQUENESS (distinct continuations ⇒ refuse), which
|
|
474
|
-
// is exactly what 4,300 competing values must trip.
|
|
475
|
-
//
|
|
476
|
-
//
|
|
477
|
-
// purpose — a candidate differing by case or punctuation
|
|
478
|
-
// capital of france" → "What is the capital of France?") is
|
|
479
|
-
// prefix, keeps grounding here, and is unaffected.
|
|
472
|
+
// is exactly what 4,300 competing values must trip. So this is not a new
|
|
473
|
+
// rule and not a new threshold: it is deferring a prefix decision to the
|
|
474
|
+
// tier that owns it (match-project.md, one factored machinery).
|
|
475
|
+
// Byte-strict on purpose — a candidate differing by case or punctuation
|
|
476
|
+
// ("what is the capital of france" → "What is the capital of France?") is
|
|
477
|
+
// NOT a byte prefix, keeps grounding here, and is unaffected.
|
|
480
478
|
const strictPrefix = g !== null &&
|
|
481
479
|
cBytes.length > query.length &&
|
|
482
480
|
indexOf(cBytes, query, 0) === 0;
|
|
@@ -542,15 +540,15 @@ export async function recallByResonance(
|
|
|
542
540
|
}
|
|
543
541
|
}
|
|
544
542
|
|
|
545
|
-
// The refusal/echo decision.
|
|
546
|
-
//
|
|
543
|
+
// The refusal/echo decision. The echo returns a stored form's bytes AS the
|
|
544
|
+
// answer — a near-identity claim about the query — and identity-grade
|
|
547
545
|
// decisions are never made on an estimated score ("approximate scores may
|
|
548
|
-
// rank and propose; they may never decide",
|
|
549
|
-
// overshooting the reach bar echoed a WRONG-entity neighbour
|
|
550
|
-
// Zamunda?" echoed the Armenia fact, observed).
|
|
551
|
-
// anyway to be echoed, so the decision uses their EXACT fold: one river
|
|
552
|
-
// fold of the top hit, measured in the same query-relative,
|
|
553
|
-
//
|
|
546
|
+
// rank and propose; they may never decide", exact-vs-approximate.md): the
|
|
547
|
+
// RaBitQ estimate overshooting the reach bar echoed a WRONG-entity neighbour
|
|
548
|
+
// ("capital of Zamunda?" echoed the Armenia fact, observed). The bytes are
|
|
549
|
+
// read anyway to be echoed, so the decision uses their EXACT fold: one river
|
|
550
|
+
// fold of the top hit, measured in the same query-relative, chance-corrected
|
|
551
|
+
// units as the tier above.
|
|
554
552
|
const topBytes = read(ctx, top.id);
|
|
555
553
|
const exact = topBytes.length > 0
|
|
556
554
|
? cosine(queryGist, gistOf(ctx, topBytes))
|
|
@@ -2,8 +2,8 @@
|
|
|
2
2
|
// bytes (Grounding IV).
|
|
3
3
|
//
|
|
4
4
|
// This file is a CONFIGURATION of the shared frame reading in match.ts, not a
|
|
5
|
-
// pipeline of its own.
|
|
6
|
-
// them and are reachable by any mechanism:
|
|
5
|
+
// pipeline of its own. The three parts it configures live where
|
|
6
|
+
// match-project.md puts them and are reachable by any mechanism:
|
|
7
7
|
//
|
|
8
8
|
// matcher Precomputed.frames() — the frame INVENTORY: which ranked
|
|
9
9
|
// candidates read as instances of the query's own frame, and
|
|
@@ -29,10 +29,10 @@
|
|
|
29
29
|
// candidate's continuation UNSUBSTITUTED, so admitting a slot-gap there would
|
|
30
30
|
// voice the corpus's filler for the asker's referent — the misreference
|
|
31
31
|
// measured live on the trained store ("How do you say 'flurbish' in French?"
|
|
32
|
-
// answered "the way to say hello is \"Bonjour\"").
|
|
32
|
+
// answered "the way to say hello is \"Bonjour\""). Nor is CAST rewired: its
|
|
33
33
|
// frame gate is WEAVE-local while a slot is COHORT-local, and substituting one
|
|
34
|
-
// population for the other is the error
|
|
35
|
-
// AVAILABLE, never imposed.
|
|
34
|
+
// population for the other is the error commonality.md names. The notion is
|
|
35
|
+
// made AVAILABLE, never imposed.
|
|
36
36
|
|
|
37
37
|
import type { MindContext } from "../types.js";
|
|
38
38
|
import type { FrameInstance } from "../match.js";
|
|
@@ -52,16 +52,16 @@ import { rItem, rNode, traceFail } from "../trace.js";
|
|
|
52
52
|
* agrees with nothing, so no carriage is attested — the same "two or no
|
|
53
53
|
* constituent" reading frame-filler's contentRuns applies.
|
|
54
54
|
*
|
|
55
|
-
* THIS IS ALSO THE MECHANISM'S REACH.
|
|
56
|
-
*
|
|
57
|
-
*
|
|
58
|
-
*
|
|
59
|
-
*
|
|
60
|
-
*
|
|
61
|
-
*
|
|
62
|
-
*
|
|
63
|
-
*
|
|
64
|
-
*
|
|
55
|
+
* THIS IS ALSO THE MECHANISM'S REACH. Evidence comes from the shared top-k
|
|
56
|
+
* resonance, so a frame the corpus instantiates only ONCE within k is not
|
|
57
|
+
* reachable here. Measured on the trained store: `How do you say 'flurbish' in
|
|
58
|
+
* French?` finds one instance of its frame in the top 24 — the rest are `How do
|
|
59
|
+
* you make …`, a different frame — so this abstains and recall's
|
|
60
|
+
* scaffolding-dominated tier answers with the CORPUS's filler. That
|
|
61
|
+
* misreference is recall's, and widening the supply is not the fix: the
|
|
62
|
+
* exhaustive √N list recall's refusal path builds costs hundreds of
|
|
63
|
+
* milliseconds and this runs before it. Abstaining on thin evidence is the
|
|
64
|
+
* honest reading (INVARIANTS.md). */
|
|
65
65
|
const MIN_INSTANCES = 2;
|
|
66
66
|
|
|
67
67
|
/** THE VOICING GATES — this mechanism's own reading of a pairing, applied here
|
|
@@ -135,7 +135,7 @@ function electFrame(
|
|
|
135
135
|
let best: FrameInstance[] = [];
|
|
136
136
|
for (const group of bySignature.values()) {
|
|
137
137
|
// Ties keep the FIRST group in insertion order, which is resonance rank —
|
|
138
|
-
// corpus-determined, like every other tie-break here (
|
|
138
|
+
// corpus-determined, like every other tie-break here (determinism.md).
|
|
139
139
|
if (group.length > best.length) best = group;
|
|
140
140
|
}
|
|
141
141
|
return best;
|
package/src/mind/mind.ts
CHANGED
|
@@ -211,13 +211,12 @@ export interface MindOptions {
|
|
|
211
211
|
host: import("../extension.js").ExtensionHost,
|
|
212
212
|
) => import("./pipeline-mechanism.js").PipelineMechanism)[];
|
|
213
213
|
/** Measure the computational usage of every inference call — see
|
|
214
|
-
* src/meter.ts.
|
|
215
|
-
*
|
|
216
|
-
*
|
|
217
|
-
*
|
|
218
|
-
*
|
|
219
|
-
*
|
|
220
|
-
* AGENTS §2.11), so profile without a trace. */
|
|
214
|
+
* src/meter.ts. Off by default and free when off (one null check per store
|
|
215
|
+
* read); on, each `respond`/`respondTurn` leaves a {@link Mind.lastCost}
|
|
216
|
+
* report behind. Counters are deterministic, so two runs of the same query on
|
|
217
|
+
* the same store are diffable; the millisecond fields are not. Profiling
|
|
218
|
+
* NEVER changes an answer — but attaching a RATIONALE does: a traced response
|
|
219
|
+
* bypasses the ctx memos (memoization.md), so profile without a trace. */
|
|
221
220
|
profile?: boolean;
|
|
222
221
|
/** Content canonicalizer applied to EVERY response (any modality) for
|
|
223
222
|
* equivalence-class resolution — see src/canon.ts. Text entry points
|
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
// a list of PipelineMechanism objects — it never imports a mechanism-specific
|
|
6
6
|
// type and never has a special-case branch for any mechanism.
|
|
7
7
|
//
|
|
8
|
-
// The four constraints of the free-will architecture (
|
|
8
|
+
// The four constraints of the free-will architecture (mechanism-market.md):
|
|
9
9
|
// 1. DECOUPLING — mechanisms import nothing from each other or from pipeline.
|
|
10
10
|
// 2. DECLARED COMPETENCE — floor() returns null when impossible, a number when
|
|
11
11
|
// possible. Binary, auditable, no learned scores.
|
|
@@ -148,17 +148,17 @@ export class Precomputed {
|
|
|
148
148
|
);
|
|
149
149
|
}
|
|
150
150
|
|
|
151
|
-
// REMOVED — the WIDE exhaustive-√N resonance list (`wideResonance`).
|
|
151
|
+
// REMOVED — the WIDE exhaustive-√N resonance list (`wideResonance`). It ran
|
|
152
152
|
// `resonate(guide, √N, exhaustive=true)` whenever the top hit cleared
|
|
153
|
-
// conceptThreshold, so consumers could look "past the top-k".
|
|
153
|
+
// conceptThreshold, so consumers could look "past the top-k". Every consumer
|
|
154
154
|
// only ever needed ≤ 2·recallQueryK proposals (the substitution bridge's own
|
|
155
155
|
// candidate cap) or a content-addressed answer (prefix completion's
|
|
156
|
-
// formsOpenedBy), and every proposal is byte-verified downstream
|
|
157
|
-
// the exhaustive scan bought recall at O(index)
|
|
158
|
-
// measured: 244K annVectorReads per refusing query,
|
|
159
|
-
// byte-identical to a top-k read.
|
|
160
|
-
// (the one top-k read) and the write side's window index
|
|
161
|
-
// recall.ts and prefix-completion.ts.
|
|
156
|
+
// formsOpenedBy), and every proposal is byte-verified downstream
|
|
157
|
+
// (exact-vs-approximate.md), so the exhaustive scan bought recall at O(index)
|
|
158
|
+
// cost for an O(k) need — measured: 244K annVectorReads per refusing query,
|
|
159
|
+
// ~1.5 s, every answer byte-identical to a top-k read. The two consumers now
|
|
160
|
+
// read `resonance()` (the one top-k read) and the write side's window index
|
|
161
|
+
// respectively — see recall.ts and prefix-completion.ts.
|
|
162
162
|
|
|
163
163
|
private _frames?: Promise<ReadonlyArray<FrameInstance>>;
|
|
164
164
|
/** THE FRAME INVENTORY — every ranked candidate that reads as an instance of
|
|
@@ -166,14 +166,16 @@ export class Precomputed {
|
|
|
166
166
|
* ({@link FrameInstance}). The one place the engine represents "a position
|
|
167
167
|
* whose occupant comes from the context rather than the corpus".
|
|
168
168
|
*
|
|
169
|
-
*
|
|
170
|
-
*
|
|
171
|
-
*
|
|
172
|
-
*
|
|
173
|
-
*
|
|
174
|
-
*
|
|
175
|
-
*
|
|
176
|
-
*
|
|
169
|
+
* AN INVENTORY, NOT AN ELECTION. It reports every pairing and elects no
|
|
170
|
+
* frame,
|
|
171
|
+
* deliberately: a slot is a property of a PAIRING, not of the query, and
|
|
172
|
+
* different candidates put slots in different places. Committing to one
|
|
173
|
+
* reading here would push whichever consumer asked first onto everyone else —
|
|
174
|
+
* the market's decoupling (mechanism-market.md) broken from inside the shared
|
|
175
|
+
* container, and the population error commonality.md names. Each consumer
|
|
176
|
+
* groups and commits for its own question; reference elects the modal slot
|
|
177
|
+
* signature, and a consumer wanting a different reading is not fighting this
|
|
178
|
+
* one.
|
|
177
179
|
*
|
|
178
180
|
* NO LICENCE EITHER. Knowing a span is variable is safe for every consumer
|
|
179
181
|
* — it can only improve an alignment. Knowing one may be VOICED through is
|
|
@@ -189,9 +191,10 @@ export class Precomputed {
|
|
|
189
191
|
const capBytes = this.query.length * W;
|
|
190
192
|
const out: FrameInstance[] = [];
|
|
191
193
|
for (const h of await this.resonance()) {
|
|
192
|
-
// REJECT BY LENGTH BEFORE RECONSTRUCTING (
|
|
193
|
-
// indexed read, `bytesPrefix` rebuilds a subtree.
|
|
194
|
-
// cap is applied — it is a bounded-read
|
|
194
|
+
// REJECT BY LENGTH BEFORE RECONSTRUCTING (bounded-reads.md):
|
|
195
|
+
// `contentLen` is an indexed read, `bytesPrefix` rebuilds a subtree.
|
|
196
|
+
// ONLY the phrase-scale cap is applied — it is a bounded-read
|
|
197
|
+
// discipline, not a judgement.
|
|
195
198
|
//
|
|
196
199
|
// A LOWER bound was here too (`dominates(len, query.length)`, on the
|
|
197
200
|
// reasoning that a candidate shorter than half the query cannot supply
|
|
@@ -519,7 +522,8 @@ function computeWeave(
|
|
|
519
522
|
// IDF — gates the aligner has no equivalent of.
|
|
520
523
|
//
|
|
521
524
|
// So the climb PROPOSES the pairing (which structure, which query span) and
|
|
522
|
-
// bytes DECIDE its terms (
|
|
525
|
+
// bytes DECIDE its terms (exact-vs-approximate.md). Three gates, each one
|
|
526
|
+
// measured:
|
|
523
527
|
//
|
|
524
528
|
// • it may only take query bytes NO literal run claimed. Run inline with
|
|
525
529
|
// phase 1 this did the opposite of "exact decides" — a higher-ranked
|
package/src/mind/pipeline.ts
CHANGED
|
@@ -130,15 +130,15 @@ export interface NarrowDecisionData {
|
|
|
130
130
|
}
|
|
131
131
|
|
|
132
132
|
/** Structured payload of the "regimePrediction" rationale step — the R8
|
|
133
|
-
* observation exposed as data.
|
|
134
|
-
*
|
|
135
|
-
*
|
|
136
|
-
*
|
|
137
|
-
*
|
|
138
|
-
*
|
|
139
|
-
*
|
|
140
|
-
*
|
|
141
|
-
*
|
|
133
|
+
* observation exposed as data. After the first mechanism (cover, which
|
|
134
|
+
* mechanism-market.md runs first) grounds or abstains, the market's whole
|
|
135
|
+
* outcome is already determined by the one cost ladder: the consensus climb
|
|
136
|
+
* runs exactly when `worthRunning(2 * STEP)` is true — CAST (floor 2·STEP) is
|
|
137
|
+
* the cheapest mechanism that first-touches it, and confluence (3·STEP) /
|
|
138
|
+
* extraction (CONCEPT+STEP) are only reached after CAST is. An incumbent at or
|
|
139
|
+
* below that floor prunes CAST and, with it, the climb (retrieval); anything
|
|
140
|
+
* above — or no incumbent — runs the full market and the climb (composition).
|
|
141
|
+
* Purely observational; never read by inference. */
|
|
142
142
|
export interface RegimePredictionData {
|
|
143
143
|
version: 1;
|
|
144
144
|
/** retrieval | composition — the two regimes R1 measured as a ~100× cost
|
|
@@ -187,13 +187,13 @@ export async function think(
|
|
|
187
187
|
// ── Pre-computation ──────────────────────────────────────────────────
|
|
188
188
|
const mechanisms = mechs ?? defaultMechanisms;
|
|
189
189
|
const meter = ctx.meter;
|
|
190
|
-
// recognition is a shared analysis (
|
|
190
|
+
// recognition is a shared analysis (meter.md contract 5): it does the query's
|
|
191
191
|
// own store work (perceive → foldTree → resolve), which used to land in
|
|
192
192
|
// `think` and in nothing narrower — the meter's one accounting surface must
|
|
193
193
|
// charge it to itself, exactly as attention/weave/resonance are charged.
|
|
194
|
-
// SYNCHRONOUS phase: recognition is on the sync side of
|
|
195
|
-
// is timed with `timeSync` — wrapping it in a promise would make a
|
|
196
|
-
// response await where an unprofiled one does not.
|
|
194
|
+
// SYNCHRONOUS phase: recognition is on the sync side of meter.md's seam, so
|
|
195
|
+
// it is timed with `timeSync` — wrapping it in a promise would make a
|
|
196
|
+
// profiled response await where an unprofiled one does not.
|
|
197
197
|
const rec = meter
|
|
198
198
|
? meter.timeSync("recognise", () => recognise(ctx, query))
|
|
199
199
|
: recognise(ctx, query);
|
|
@@ -225,15 +225,15 @@ export async function think(
|
|
|
225
225
|
}
|
|
226
226
|
}
|
|
227
227
|
|
|
228
|
-
// Phase 2: the shared pre-computation container.
|
|
229
|
-
// (recognition, computed spans, guide) — every expensive analysis
|
|
230
|
-
//
|
|
231
|
-
//
|
|
232
|
-
//
|
|
233
|
-
//
|
|
234
|
-
//
|
|
235
|
-
//
|
|
236
|
-
//
|
|
228
|
+
// Phase 2: the shared pre-computation container. Eager fields only
|
|
229
|
+
// (recognition, computed spans, guide) — every expensive analysis (consensus
|
|
230
|
+
// climb, weave, span-shape classification) is a lazily-cached method on
|
|
231
|
+
// Precomputed, first-touched by whichever mechanism's floor survives its
|
|
232
|
+
// cheap gates and the worthRunning check. A query no mechanism climbs for
|
|
233
|
+
// (e.g. one an extension decided) never climbs. NOT phased: the constructor
|
|
234
|
+
// itself is trivial (it only derives `k`), so a phase here would add a
|
|
235
|
+
// zero-work entry to every profiled report — the meter attributes WORK
|
|
236
|
+
// (meter.md); the trace already represents structure.
|
|
237
237
|
const pre = new Precomputed(ctx, query, rec, computed, ctx._edgeGuide);
|
|
238
238
|
|
|
239
239
|
// ── Grounding: ONE lightest-derivation choice among the mechanisms ────
|
|
@@ -298,16 +298,17 @@ export async function think(
|
|
|
298
298
|
const worthRunning = (floor: number) =>
|
|
299
299
|
best === null || grade(floor) < grade(best.weight);
|
|
300
300
|
|
|
301
|
-
// REGIME PREDICTION (R8) — observational only.
|
|
302
|
-
// had its turn (cover, which
|
|
303
|
-
// outcome is already determined by the one cost ladder: the
|
|
304
|
-
// runs exactly when `worthRunning(2 * STEP)` is true — CAST
|
|
305
|
-
// the cheapest mechanism that first-touches it, so an
|
|
306
|
-
// grade 2 prunes CAST and, with it, confluence (3·STEP)
|
|
307
|
-
// (CONCEPT+STEP) (retrieval); anything above — or no incumbent
|
|
308
|
-
// full market and the climb (composition).
|
|
309
|
-
// the same function the loop itself uses — nothing is
|
|
310
|
-
// engine had not already computed, and nothing is read
|
|
301
|
+
// REGIME PREDICTION (R8) — observational only. Once the FIRST mechanism has
|
|
302
|
+
// had its turn (cover, which mechanism-market.md places first and floors at
|
|
303
|
+
// 0), the market's outcome is already determined by the one cost ladder: the
|
|
304
|
+
// consensus climb runs exactly when `worthRunning(2 * STEP)` is true — CAST
|
|
305
|
+
// (floor 2·STEP) is the cheapest mechanism that first-touches it, so an
|
|
306
|
+
// incumbent at or below grade 2 prunes CAST and, with it, confluence (3·STEP)
|
|
307
|
+
// and extraction (CONCEPT+STEP) (retrieval); anything above — or no incumbent
|
|
308
|
+
// — runs the full market and the climb (composition). The predicate is
|
|
309
|
+
// `worthRunning`, the same function the loop itself uses — nothing is
|
|
310
|
+
// computed here that the engine had not already computed, and nothing is read
|
|
311
|
+
// back by inference.
|
|
311
312
|
//
|
|
312
313
|
// EMITTED BEFORE THE SECOND MECHANISM'S FLOOR, never after some mechanism's
|
|
313
314
|
// run: a "prediction" published after the fact could assert "the climb will
|
package/src/mind/primitives.ts
CHANGED
|
@@ -59,11 +59,11 @@ export function perceiveKey(
|
|
|
59
59
|
/** Perceive input into a content-defined tree (the river fold).
|
|
60
60
|
* Deterministic — identical bytes always produce an identical tree.
|
|
61
61
|
*
|
|
62
|
-
*
|
|
63
|
-
*
|
|
64
|
-
*
|
|
65
|
-
*
|
|
66
|
-
*
|
|
62
|
+
* `boundaries` is an optional sorted list of proper byte offsets where the fold
|
|
63
|
+
* must split so that each prefix segment folds identically to how it folded
|
|
64
|
+
* when it was learned (fold-contract.md stable-prefix contract). Only the
|
|
65
|
+
* CALLER — who assembled the multi-turn context — knows where those boundaries
|
|
66
|
+
* are; the geometry never guesses them from the bytes. */
|
|
67
67
|
export function perceive(
|
|
68
68
|
ctx: MindContext,
|
|
69
69
|
input: Input,
|
package/src/mind/recognition.ts
CHANGED
|
@@ -34,19 +34,20 @@ import type { Leaf, Site } from "./graph-search.js";
|
|
|
34
34
|
*
|
|
35
35
|
* Both O(n · maxGroup) bounded O(1) probes — never a scan of the corpus.
|
|
36
36
|
*
|
|
37
|
-
*
|
|
38
|
-
*
|
|
39
|
-
*
|
|
40
|
-
*
|
|
41
|
-
*
|
|
42
|
-
*
|
|
43
|
-
*
|
|
44
|
-
*
|
|
45
|
-
*
|
|
46
|
-
*
|
|
47
|
-
*
|
|
48
|
-
*
|
|
49
|
-
*
|
|
37
|
+
* ONE READING PER BYTE STREAM, deliberately: there is no "cheap mode" that
|
|
38
|
+
* skips
|
|
39
|
+
* the edge-trim fallbacks. A `trimmed` variant was tried and REFUTED twice
|
|
40
|
+
* over. Its premise — "the trims only recover misaligned FRAGMENTS, so a
|
|
41
|
+
* consumer whose gate rejects fragments loses nothing" — is false: the
|
|
42
|
+
* left/right trim loops below exist precisely to find WHOLE trained forms
|
|
43
|
+
* embedded at an offset the query's own fold did not cut, and such a form has
|
|
44
|
+
* no structural parents or containers, so it passes the pivot's fragment gate
|
|
45
|
+
* and is exactly the candidate a multi-hop chain steps through. Skipping them
|
|
46
|
+
* narrows the pivot's evidence silently. And a per-caller variant has to key
|
|
47
|
+
* the memo by the variant, which breaks the "computed at most once" property
|
|
48
|
+
* (memoization.md): the pipeline recognises a grounded answer untrimmed for
|
|
49
|
+
* `preConsumed`, and the pivot then recognises the same bytes again — the
|
|
50
|
+
* saving inverts into a doubling on the path it was measured for. */
|
|
50
51
|
export function recognise(
|
|
51
52
|
ctx: MindContext,
|
|
52
53
|
bytes: Uint8Array,
|
|
@@ -180,16 +181,15 @@ function recogniseImpl(ctx: MindContext, bytes: Uint8Array): Recognition {
|
|
|
180
181
|
// and read the answer from `starts`, which is exactly {0, W, 2W, …}
|
|
181
182
|
// because riverFold groups fixed-arity — arithmetic, not evidence.
|
|
182
183
|
//
|
|
183
|
-
// Measured on the 17.9M-node store, over the sites of 7 probes (1 good,
|
|
184
|
-
//
|
|
185
|
-
//
|
|
186
|
-
//
|
|
187
|
-
//
|
|
188
|
-
//
|
|
189
|
-
//
|
|
190
|
-
//
|
|
191
|
-
//
|
|
192
|
-
// rarity does not separate: "hi" has 1 container, "the" 572
|
|
184
|
+
// Measured on the 17.9M-node store, over the sites of 7 probes (1 good, 11
|
|
185
|
+
// junk by hand-labelling, corrected for whole-query forms): len >= W
|
|
186
|
+
// rejects "hi"(2) "of"(2) "is"(2) "di"(2) "the"(3), admits "Eiffel
|
|
187
|
+
// Tower"(12) and both whole-query forms len >= W-1 admits "the" — W-1 is
|
|
188
|
+
// the write side's straddle neighbour for RETRIEVAL, never a claim about
|
|
189
|
+
// units commonality.md saturation admits 11/11 junk: edgeAncestors on a
|
|
190
|
+
// site node reaches 1..48 contexts, so dominates(ctx, N) needs ctx > 162805
|
|
191
|
+
// and never fires; every site reads DISC rarity does not separate: "hi" has
|
|
192
|
+
// 1 container, "the" 572
|
|
193
193
|
//
|
|
194
194
|
// A span covering the WHOLE query is exempt: then it is not a fragment of
|
|
195
195
|
// something longer, it is the question ("hi" asked on its own).
|
|
@@ -517,7 +517,7 @@ function recogniseImpl(ctx: MindContext, bytes: Uint8Array): Recognition {
|
|
|
517
517
|
start: number,
|
|
518
518
|
end: number,
|
|
519
519
|
canonBudget: boolean,
|
|
520
|
-
):
|
|
520
|
+
): boolean => {
|
|
521
521
|
// Any span at least one river window wide is worth a probe. This used
|
|
522
522
|
// to stop at `chainReach(W)` — "the chain already covers anything that
|
|
523
523
|
// short" — and that premise does not hold for every embedded form: the
|
|
@@ -531,14 +531,17 @@ function recogniseImpl(ctx: MindContext, bytes: Uint8Array): Recognition {
|
|
|
531
531
|
// starts on a fold cut nor ends on a node edge was unreachable by either
|
|
532
532
|
// tier — the exact site whose loss `tryChain`'s own note records as "the
|
|
533
533
|
// pivot dies with the site and multi-hop goes silent". The interior
|
|
534
|
-
// pass below spends the same budget on those pairs.
|
|
535
|
-
|
|
534
|
+
// pass below spends the same budget on those pairs. Returns whether it
|
|
535
|
+
// emitted, so a caller can retry a trimmed edge on the miss path only.
|
|
536
|
+
if (end - start < W) return false;
|
|
536
537
|
if (flatProbe(start, end) === null) {
|
|
537
|
-
if (!canonBudget) return;
|
|
538
|
-
if (!canonAdmits(start, end)) return;
|
|
538
|
+
if (!canonBudget) return false;
|
|
539
|
+
if (!canonAdmits(start, end)) return false;
|
|
539
540
|
}
|
|
540
541
|
const id = resolveSpan(start, end);
|
|
541
|
-
if (id
|
|
542
|
+
if (id === null) return false;
|
|
543
|
+
emit(start, end, id);
|
|
544
|
+
return true;
|
|
542
545
|
};
|
|
543
546
|
// A CUMULATIVE BYTE BUDGET, SPENT SHORTEST-SPAN-FIRST.
|
|
544
547
|
//
|
|
@@ -574,10 +577,7 @@ function recogniseImpl(ctx: MindContext, bytes: Uint8Array): Recognition {
|
|
|
574
577
|
const span = end - start;
|
|
575
578
|
const afford = span <= budget;
|
|
576
579
|
if (afford) budget -= span;
|
|
577
|
-
probe(start, end, afford);
|
|
578
|
-
// Always keep walking: the exact route is unbudgeted, so running out
|
|
579
|
-
// of canon budget must not stop the scan.
|
|
580
|
-
return true;
|
|
580
|
+
return probe(start, end, afford);
|
|
581
581
|
};
|
|
582
582
|
const prefixes = ordered.filter((e) => e > 0).sort((a, b) => a - b);
|
|
583
583
|
const suffixes = ordered
|
|
@@ -585,9 +585,24 @@ function recogniseImpl(ctx: MindContext, bytes: Uint8Array): Recognition {
|
|
|
585
585
|
.sort((a, b) => b - a);
|
|
586
586
|
for (let i = 0; i < Math.max(prefixes.length, suffixes.length); i++) {
|
|
587
587
|
// Interleaved so neither edge starves the other when the budget runs
|
|
588
|
-
// out — a query can carry a trained form at either end.
|
|
589
|
-
|
|
590
|
-
|
|
588
|
+
// out — a query can carry a trained form at either end. The scan never
|
|
589
|
+
// stops on a miss: the exact route is unbudgeted, so running out of
|
|
590
|
+
// canon budget must not stop it.
|
|
591
|
+
if (i < prefixes.length) spend(0, prefixes[i]);
|
|
592
|
+
if (i < suffixes.length) {
|
|
593
|
+
const s = suffixes[i];
|
|
594
|
+
// An edge form can end ONE byte before the edge does: a query often
|
|
595
|
+
// closes with a separator ("…is Timur Bekmambetov.") that belongs
|
|
596
|
+
// BETWEEN forms, and the text canonicalizer passes punctuation through,
|
|
597
|
+
// so the full-edge probe can never match it. Retry the trimmed edge
|
|
598
|
+
// ON THE MISS PATH ONLY — self-verifying (resolve decides, so a wrong
|
|
599
|
+
// trim can never emit), the same ±1 edge-trim discipline the canon-miss
|
|
600
|
+
// fallback above already trusts, and the hit path pays nothing. A form
|
|
601
|
+
// LONGER than `chainReach` at the sentence end (measured: "Timur
|
|
602
|
+
// Bekmambetov", 17 bytes) is recovered only here — the interior pass is
|
|
603
|
+
// capped at `chainReach`.
|
|
604
|
+
if (!spend(s, bytes.length)) spend(s, bytes.length - 1);
|
|
605
|
+
}
|
|
591
606
|
}
|
|
592
607
|
// INTERIOR pairs within the same `chainReach(W)` span bound the chain
|
|
593
608
|
// trusts — the dead zone the gate above used to leave: a form that neither
|
package/src/mind/resonance.ts
CHANGED
|
@@ -380,15 +380,15 @@ export async function pivotInto(
|
|
|
380
380
|
// Byte containment, longest wins — the answer literally contains the
|
|
381
381
|
// pivot's bytes, and the biggest well-evidenced span is the real pivot.
|
|
382
382
|
//
|
|
383
|
-
// REAL SATURATION, not a hard cap: the score IS the candidate's byte
|
|
384
|
-
//
|
|
385
|
-
//
|
|
386
|
-
//
|
|
387
|
-
// the cheap ordering key, and the first-inserted tie-break is made
|
|
388
|
-
// (`a.index - b.index`) so equal lengths keep `scored`'s insertion
|
|
389
|
-
// exactly the tie argmaxBy(strict) used to keep.
|
|
390
|
-
// winning candidate are read; every shorter candidate the probes proposed
|
|
391
|
-
// skipped without reconstruction, where the old argmax read them all.
|
|
383
|
+
// REAL SATURATION, not a hard cap: the score IS the candidate's byte length,
|
|
384
|
+
// so the scan is DECIDED the moment the first candidate that passes every
|
|
385
|
+
// filter is found in DESCENDING length order — a shorter candidate can never
|
|
386
|
+
// outscore it. `contentLen` (the prefix-capped length read, bounded-reads.md)
|
|
387
|
+
// is the cheap ordering key, and the first-inserted tie-break is made
|
|
388
|
+
// explicit (`a.index - b.index`) so equal lengths keep `scored`'s insertion
|
|
389
|
+
// order — exactly the tie argmaxBy(strict) used to keep. The bytes of at most
|
|
390
|
+
// ONE winning candidate are read; every shorter candidate the probes proposed
|
|
391
|
+
// is skipped without reconstruction, where the old argmax read them all.
|
|
392
392
|
const ranked = [...scored.keys()]
|
|
393
393
|
.map((id, index) => ({
|
|
394
394
|
id,
|
|
@@ -399,11 +399,11 @@ export async function pivotInto(
|
|
|
399
399
|
let pivotId: number | null = null;
|
|
400
400
|
for (const c of ranked) {
|
|
401
401
|
const id = c.id;
|
|
402
|
-
// A ZERO-LENGTH candidate is not a pivot.
|
|
402
|
+
// A ZERO-LENGTH candidate is not a pivot. `argmaxBy(…, 0, strict)` used to
|
|
403
403
|
// carry this floor in its threshold argument, and dropping it here would
|
|
404
404
|
// admit an empty node: `indexOf(answer, <empty>)` returns 0, so every
|
|
405
|
-
// filter below passes and the chain would hop through nothing
|
|
406
|
-
// empty bytes are truthy).
|
|
405
|
+
// filter below passes and the chain would hop through nothing
|
|
406
|
+
// (INVARIANTS.md — empty bytes are truthy).
|
|
407
407
|
if (c.len === 0) continue;
|
|
408
408
|
// A PIVOT MUST BE A THING THE CORPUS DEPOSITED, NOT A PIECE OF ONE.
|
|
409
409
|
// "Longest wins" ranks candidates but never asks whether the winner is
|
|
@@ -439,15 +439,15 @@ export async function pivotInto(
|
|
|
439
439
|
// a span that was never a fact on its own is not one to step through.
|
|
440
440
|
// No constant enters — it is a structural predicate, not a threshold.
|
|
441
441
|
if (ctx.store.hasParents(id) || ctx.store.hasContainers(id)) continue;
|
|
442
|
-
// A candidate whose bytes are LONGER than the answer cannot be a
|
|
443
|
-
//
|
|
444
|
-
//
|
|
445
|
-
//
|
|
446
|
-
//
|
|
447
|
-
//
|
|
448
|
-
//
|
|
449
|
-
//
|
|
450
|
-
//
|
|
442
|
+
// A candidate whose bytes are LONGER than the answer cannot be a substring
|
|
443
|
+
// of it — `indexOf` would return −1 regardless. Prune by length BEFORE
|
|
444
|
+
// reconstructing the bytes: `read` is an UNCAPPED read (bounded-reads.md),
|
|
445
|
+
// and a resonated context far longer than the answer is exactly the
|
|
446
|
+
// candidate that makes it cost a whole deposit's worth of reconstruction
|
|
447
|
+
// for a containment test that must fail. `contentLen` with the
|
|
448
|
+
// `answer.length + 1` cap is the prefix-capped length read the same
|
|
449
|
+
// contract prescribes; the prune is byte-identical to the old `indexOf`
|
|
450
|
+
// miss (it returns −1 for a needle longer than the haystack).
|
|
451
451
|
if (c.len > answer.length) continue;
|
|
452
452
|
const bytes = read(ctx, id);
|
|
453
453
|
if (indexOf(answer, bytes, 0) < 0) continue;
|