@hviana/sema 0.8.0 → 0.8.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +22 -1
- package/DATASETS.md +1 -1
- package/dist/example/train_base/config.js +2 -2
- package/dist/example/train_base/corpora/massive.js +1 -1
- package/dist/example/train_base/readers.js +1 -1
- package/dist/src/config.d.ts +17 -0
- package/dist/src/config.js +18 -0
- package/dist/src/geometry.d.ts +10 -10
- package/dist/src/geometry.js +25 -24
- package/dist/src/meter.d.ts +29 -12
- package/dist/src/meter.js +58 -14
- package/dist/src/mind/attention.js +12 -12
- package/dist/src/mind/bridge.d.ts +8 -8
- package/dist/src/mind/bridge.js +33 -32
- package/dist/src/mind/corpus.d.ts +40 -0
- package/dist/src/mind/corpus.js +149 -0
- package/dist/src/mind/graph-search.d.ts +7 -8
- package/dist/src/mind/graph-search.js +244 -32
- package/dist/src/mind/index.d.ts +3 -1
- package/dist/src/mind/index.js +1 -0
- package/dist/src/mind/junction.d.ts +1 -1
- package/dist/src/mind/junction.js +8 -8
- package/dist/src/mind/learning.js +36 -35
- package/dist/src/mind/match.d.ts +8 -3
- package/dist/src/mind/match.js +156 -71
- package/dist/src/mind/mechanisms/cast.js +18 -2
- package/dist/src/mind/mechanisms/cover.js +19 -12
- package/dist/src/mind/mechanisms/prefix-completion.js +24 -24
- package/dist/src/mind/mechanisms/recall.js +38 -40
- package/dist/src/mind/mechanisms/reference.js +16 -16
- package/dist/src/mind/mind.d.ts +61 -7
- package/dist/src/mind/mind.js +72 -2
- package/dist/src/mind/pipeline-mechanism.d.ts +10 -8
- package/dist/src/mind/pipeline-mechanism.js +25 -21
- package/dist/src/mind/pipeline.d.ts +9 -9
- package/dist/src/mind/pipeline.js +49 -29
- package/dist/src/mind/primitives.d.ts +5 -5
- package/dist/src/mind/primitives.js +5 -5
- package/dist/src/mind/reasoning.d.ts +5 -1
- package/dist/src/mind/reasoning.js +54 -1
- package/dist/src/mind/recognition.d.ts +14 -13
- package/dist/src/mind/recognition.js +23 -23
- package/dist/src/mind/resonance.js +21 -21
- package/dist/src/mind/traverse.d.ts +54 -52
- package/dist/src/mind/traverse.js +83 -73
- package/dist/src/mind/types.d.ts +26 -4
- package/dist/src/store.d.ts +12 -12
- package/dist/src/store.js +12 -12
- package/docs/INDEX.md +2 -2
- package/docs/architecture/exact-vs-approximate.md +2 -1
- package/docs/architecture/fold-contract.md +1 -1
- package/docs/failures/tempting-but-wrong.md +33 -5
- package/docs/harness/gates.md +7 -7
- package/example/train_base/config.ts +2 -2
- package/example/train_base/corpora/massive.ts +1 -1
- package/example/train_base/readers.ts +1 -1
- package/jsr.json +1 -1
- package/package.json +1 -1
- package/src/config.ts +35 -0
- package/src/geometry.ts +25 -24
- package/src/meter.ts +61 -14
- package/src/mind/attention.ts +12 -12
- package/src/mind/bridge.ts +33 -32
- package/src/mind/corpus.ts +202 -0
- package/src/mind/graph-search.ts +261 -31
- package/src/mind/index.ts +8 -1
- package/src/mind/junction.ts +8 -8
- package/src/mind/learning.ts +36 -35
- package/src/mind/match.ts +163 -73
- package/src/mind/mechanisms/cast.ts +17 -1
- package/src/mind/mechanisms/cover.ts +18 -12
- package/src/mind/mechanisms/prefix-completion.ts +24 -24
- package/src/mind/mechanisms/recall.ts +38 -40
- package/src/mind/mechanisms/reference.ts +16 -16
- package/src/mind/mind.ts +129 -7
- package/src/mind/pipeline-mechanism.ts +25 -21
- package/src/mind/pipeline.ts +63 -38
- package/src/mind/primitives.ts +5 -5
- package/src/mind/reasoning.ts +55 -0
- package/src/mind/recognition.ts +23 -23
- package/src/mind/resonance.ts +21 -21
- package/src/mind/traverse.ts +83 -73
- package/src/mind/types.ts +30 -4
- package/src/store.ts +20 -20
- package/test/08-storage.test.mjs +1 -1
- package/test/100-complete-grounding-trace.test.mjs +109 -0
- package/test/101-alignment-gap-bound.test.mjs +106 -0
- package/test/102-production-composes-at-scale.test.mjs +110 -0
- package/test/103-alignment-gap-budget.test.mjs +89 -0
- package/test/104-composition-is-reported.test.mjs +90 -0
- package/test/105-derive-through-reports-its-refusal.test.mjs +113 -0
- package/test/106-the-join-fires.test.mjs +94 -0
- package/test/107-the-join-is-counted.test.mjs +81 -0
- package/test/108-the-join-chains.test.mjs +78 -0
- package/test/109-the-pivot-is-counted.test.mjs +60 -0
- package/test/110-the-reasoner-stops-when-the-question-is-answered.test.mjs +91 -0
- package/test/111-the-cover-assembly-is-counted.test.mjs +74 -0
- package/test/112-the-exploration-does-not-grow-with-the-hub.test.mjs +89 -0
- package/test/113-the-rationale-payload-is-bounded.test.mjs +84 -0
- package/test/114-alignment-budget-is-per-sweep.test.mjs +93 -0
- package/test/116-the-extension-is-gated-by-the-pipelines-own-remainder.test.mjs +100 -0
- package/test/117-corpus-search.test.mjs +171 -0
- package/test/14-scaling.test.mjs +10 -7
- package/test/35-prefix-edge.test.mjs +1 -1
- package/test/40-choosenext-scale-guard.test.mjs +16 -17
- package/test/56-bridge-identity-admission.test.mjs +6 -6
- package/test/70-prefix-completion.test.mjs +4 -3
- package/test/72-prefix-candidate-supply.test.mjs +3 -3
- package/test/73-scaffolding-only-bridge-abstains.test.mjs +6 -6
- package/test/75-multiturn-context-optimisation.test.mjs +5 -5
- package/test/76-reference-binding.test.mjs +6 -1
- package/test/84-composed-answer-honesty.test.mjs +5 -6
- package/test/88-dependency-footprint.test.mjs +1 -1
- package/test/89-completion-recursion.test.mjs +47 -19
- package/test/90-connector-read-cap.test.mjs +10 -8
- package/test/93-regime-prediction.test.mjs +10 -10
- package/test/94-cross-region-budget.test.mjs +2 -2
- package/test/95-wide-resonance-removed.test.mjs +8 -7
- package/test/96-bytes-walk-termination.test.mjs +3 -3
package/src/geometry.ts
CHANGED
|
@@ -349,19 +349,20 @@ function bytesToLeaves(
|
|
|
349
349
|
* sentences fall would be importing an assumption the architecture rejects.
|
|
350
350
|
* Random binary must, and does, behave exactly like prose.
|
|
351
351
|
*
|
|
352
|
-
* Every constant is derived (
|
|
353
|
-
*
|
|
354
|
-
*
|
|
355
|
-
*
|
|
356
|
-
*
|
|
357
|
-
*
|
|
358
|
-
*
|
|
359
|
-
*
|
|
360
|
-
*
|
|
361
|
-
*
|
|
362
|
-
*
|
|
363
|
-
*
|
|
364
|
-
*
|
|
352
|
+
* Every constant is derived (thresholds.md): the cut mask is W, so a cut is
|
|
353
|
+
* offered once per quantum of bytes — which, composed with the minimum below,
|
|
354
|
+
* puts the expected segment at minLen + W − 1 ≈ 6 B rather than at W,
|
|
355
|
+
* deliberately (see the refutation recorded at `cutRate` in {@link
|
|
356
|
+
* contentLevels}: a segment is the flat PHRASE-scale unit the W-ary groups are
|
|
357
|
+
* built from, not a group of W children, and forcing E[len] = W costs 15
|
|
358
|
+
* tests). The minimum is W−1, `canonicalWindows`'s straddle neighbour and the
|
|
359
|
+
* write side's own floor for a unit; and the maximum is the KEYRING's seat
|
|
360
|
+
* count, because a segment folds as ONE flat node and `fold` has exactly that
|
|
361
|
+
* many seats to bind children into. Capping there is what keeps the fold light:
|
|
362
|
+
* a segment of 3..seats leaves is a single node, where splitting it into
|
|
363
|
+
* W-groups plus a remainder would cost two or three and the remainders barely
|
|
364
|
+
* share (measured: partial-arity nodes 504 → 3,590, and total distinct nodes
|
|
365
|
+
* 8,142 → 9,712, when segments folded as [W][rest]). */
|
|
365
366
|
/** {@link contentBoundaries} plus, for each cut, its LEVEL — how deep in the
|
|
366
367
|
* tree that cut reaches.
|
|
367
368
|
*
|
|
@@ -622,16 +623,16 @@ export function knownPrefixLength(
|
|
|
622
623
|
* correct boundary. Pass them through from `perceive`; the geometry
|
|
623
624
|
* computes the stable prefix internally.
|
|
624
625
|
*
|
|
625
|
-
* `boundaries` is the CALLER-computed stable-prefix boundary set
|
|
626
|
-
*
|
|
627
|
-
*
|
|
628
|
-
*
|
|
629
|
-
*
|
|
630
|
-
*
|
|
631
|
-
*
|
|
632
|
-
*
|
|
633
|
-
*
|
|
634
|
-
*
|
|
626
|
+
* `boundaries` is the CALLER-computed stable-prefix boundary set
|
|
627
|
+
* (fold-contract.md): strictly-increasing proper byte offsets, each the length
|
|
628
|
+
* of a prefix that is already a stored whole-stream form. When given, the fold
|
|
629
|
+
* splits into the segments between consecutive boundaries — each folded
|
|
630
|
+
* independently, exactly as it folded when it was learned — and the segment
|
|
631
|
+
* roots join LEFT-NESTED (((s₀·s₁)·s₂)…), so every learnt cumulative-context
|
|
632
|
+
* root reappears as an identical subtree (and, by hash-consing, the very same
|
|
633
|
+
* node) inside the grown stream. This is what lets a conversation's next turn
|
|
634
|
+
* extend perception instead of refolding it: identical prefixes produce
|
|
635
|
+
* identical subtrees regardless of what follows them. */
|
|
635
636
|
export function bytesToTree(
|
|
636
637
|
space: Space,
|
|
637
638
|
alphabet: Alphabet,
|
|
@@ -962,7 +963,7 @@ function flatFold(
|
|
|
962
963
|
return { tree: sema(gist, null, kids), len: n };
|
|
963
964
|
}
|
|
964
965
|
|
|
965
|
-
|
|
966
|
+
/* * The stable-prefix segmented fold (fold-contract.md). Each segment between
|
|
966
967
|
* consecutive boundaries folds PLAINLY and independently; segment roots
|
|
967
968
|
* join left-nested, and only the final root is normalized (the linear-fold
|
|
968
969
|
* contract: one normalize per perception). A segment's own inner splits
|
package/src/meter.ts
CHANGED
|
@@ -8,8 +8,8 @@
|
|
|
8
8
|
// Four contracts, all load-bearing:
|
|
9
9
|
//
|
|
10
10
|
// 1. NEVER READ BY INFERENCE. No counter may reach a decision, a threshold,
|
|
11
|
-
// or an ordering. Determinism (
|
|
12
|
-
// meter is write-only from the engine's point of view.
|
|
11
|
+
// or an ordering. Determinism (determinism.md) survives
|
|
12
|
+
// only because the meter is write-only from the engine's point of view.
|
|
13
13
|
// 2. OFF BY DEFAULT, AND FREE WHEN OFF. Every call site is `meter?.x++` on
|
|
14
14
|
// a null field. Nothing allocates, nothing is keyed, nothing is timed
|
|
15
15
|
// unless a Meter is attached (`new Mind({ profile: true })`).
|
|
@@ -79,9 +79,9 @@ export class Meter {
|
|
|
79
79
|
nodeRecords = 0;
|
|
80
80
|
/** `store.bytes` / `store.bytesPrefix` — one reconstruction request. */
|
|
81
81
|
byteReads = 0;
|
|
82
|
-
|
|
83
|
-
*
|
|
84
|
-
*
|
|
82
|
+
/* * Bytes actually handed back by those reads — the real I/O volume, and the
|
|
83
|
+
* number that exposes an unbounded read (bounded-reads.md) that a call count
|
|
84
|
+
* alone hides. */
|
|
85
85
|
bytesRead = 0;
|
|
86
86
|
/** `store.contentLen`. */
|
|
87
87
|
lenReads = 0;
|
|
@@ -178,10 +178,10 @@ export class Meter {
|
|
|
178
178
|
junctionPops = 0;
|
|
179
179
|
/** Ascents that ended by EXHAUSTING the expansion budget rather than by
|
|
180
180
|
* deciding — the walk abstained and the caller silently fell through to a
|
|
181
|
-
*
|
|
182
|
-
*
|
|
183
|
-
*
|
|
184
|
-
*
|
|
181
|
+
* lower tier of the ladder — honest degradation, and nothing else reports it
|
|
182
|
+
* (INVARIANTS.md). It rises the moment a SHARED budget is drained by an
|
|
183
|
+
* earlier walk, which is what makes "this tier answered nothing"
|
|
184
|
+
* distinguishable from "this tier never got to look". */
|
|
185
185
|
junctionBudgetExhausted = 0;
|
|
186
186
|
/** Arbitrary byte spans whose distributional company was VSA-bundled from
|
|
187
187
|
* existing episode halos. */
|
|
@@ -206,6 +206,53 @@ export class Meter {
|
|
|
206
206
|
/** Candidates the decider weighed. */
|
|
207
207
|
candidates = 0;
|
|
208
208
|
|
|
209
|
+
// ── Graph search: the fact join (DIRECTION) ─────────────────────────────
|
|
210
|
+
//
|
|
211
|
+
// The join's outcome was observable ONLY through the rationale, and the
|
|
212
|
+
// rationale PERTURBS the search (measured: appending text to a refusal note
|
|
213
|
+
// changed a traced answer). These four counters are the untraced view — the
|
|
214
|
+
// same surface every other work counter uses, incremented where the decision
|
|
215
|
+
// is made, never behind a trace guard.
|
|
216
|
+
/** `deriveThrough` yielded — a fact was reached through the subject the query
|
|
217
|
+
* never named. */
|
|
218
|
+
joinFired = 0;
|
|
219
|
+
/** Refused: no key names the entity and the tail together. (A key that
|
|
220
|
+
* resolves but leads nowhere is not "refused" — it is not the relation, so
|
|
221
|
+
* the scan simply moves on; there is no counter for a case the loop cannot
|
|
222
|
+
* reach.) */
|
|
223
|
+
joinNoKey = 0;
|
|
224
|
+
/** Refused: the fact contains no entity that leads anywhere. */
|
|
225
|
+
joinNoEntity = 0;
|
|
226
|
+
|
|
227
|
+
// ── Mind: the multi-hop pivot (EXTENSION) ───────────────────────────────
|
|
228
|
+
//
|
|
229
|
+
// `pivotStep` was observable only through the rationale, and the rationale
|
|
230
|
+
// perturbs the search (measured). How far the reasoner hopped is a
|
|
231
|
+
// BEHAVIOUR, so it needs an untraced view: one counter, incremented where the
|
|
232
|
+
// step is emitted.
|
|
233
|
+
/** Times the reasoner pivoted on a span its answer contains and stepped
|
|
234
|
+
* across that fact. */
|
|
235
|
+
pivotSteps = 0;
|
|
236
|
+
|
|
237
|
+
// ── Mind: the cover's connector assembly (LIMIT) ────────────────────────
|
|
238
|
+
//
|
|
239
|
+
// The cover's `run` is 91% of a hub query's time (`"Hello."`: 2.7 s of 3.0 s)
|
|
240
|
+
// and holds its ~270 MB peak, and none of it was countable: `searchPushes`
|
|
241
|
+
// and `candidates` do not see the connector assembly. These two counters are
|
|
242
|
+
// the untraced view of it.
|
|
243
|
+
/** `bridge` calls the cover makes assembling connectors (pairwise + n-ary). */
|
|
244
|
+
coverBridges = 0;
|
|
245
|
+
/** Continuations a CHAIN hop offered the search. Bounded by the question
|
|
246
|
+
* (`ceil(queryLen / W)`) rather than by the corpus's fan-out — measured on a
|
|
247
|
+
* hub of degree 1083, offering every continuation grew the chart to 3113 outs
|
|
248
|
+
* and cost a 270 MB peak / 256 MB OOM for a two-word question. */
|
|
249
|
+
chainOffers = 0;
|
|
250
|
+
/** Σ byte-allowance the n-ary interior passes those bridges. The allowance
|
|
251
|
+
* is `middleBytes + (m + 1) * W` — every intermediate answer's bytes plus
|
|
252
|
+
* one window of glue per joint — so it is the quantity that grows with a hub
|
|
253
|
+
* query's answers, and the first thing to read when the peak moves. */
|
|
254
|
+
coverAllowanceBytes = 0;
|
|
255
|
+
|
|
209
256
|
// ── Phases ──────────────────────────────────────────────────────────────
|
|
210
257
|
|
|
211
258
|
private readonly _phases = new Map<string, PhaseCost>();
|
|
@@ -238,11 +285,11 @@ export class Meter {
|
|
|
238
285
|
}
|
|
239
286
|
}
|
|
240
287
|
|
|
241
|
-
|
|
242
|
-
*
|
|
243
|
-
*
|
|
244
|
-
*
|
|
245
|
-
*
|
|
288
|
+
/* * Time one SYNCHRONOUS phase. The sync/async seam is a real contract
|
|
289
|
+
* (meter.md) — perception, recognition and the graph search are synchronous —
|
|
290
|
+
* so a synchronous layer must not be wrapped in `time`'s promise just to be
|
|
291
|
+
* measured: that would make the profiled path await where the unprofiled one
|
|
292
|
+
* does not, and a meter never changes what a layer computes. */
|
|
246
293
|
timeSync<T>(phase: string, fn: () => T): T {
|
|
247
294
|
const before = this.snapshot();
|
|
248
295
|
const t = performance.now();
|
package/src/mind/attention.ts
CHANGED
|
@@ -1941,10 +1941,10 @@ export function canonicalChunkId(
|
|
|
1941
1941
|
// CAST lost a point of attention it needed (test/29 D1/D2).
|
|
1942
1942
|
//
|
|
1943
1943
|
// So scan every offset and prefer an anchor that still discriminates: not
|
|
1944
|
-
// saturated, and among those the one reaching the FEWEST contexts
|
|
1945
|
-
// corpus-global).
|
|
1946
|
-
// old generalising choice stand — there is then no
|
|
1947
|
-
// find, and abstaining is the honest outcome.
|
|
1944
|
+
// saturated, and among those the one reaching the FEWEST contexts
|
|
1945
|
+
// (commonality.md, corpus-global). Only when every window in the region
|
|
1946
|
+
// saturates does the old generalising choice stand — there is then no
|
|
1947
|
+
// discriminative anchor to find, and abstaining is the honest outcome.
|
|
1948
1948
|
let discId: number | null = null;
|
|
1949
1949
|
let discReached = Infinity;
|
|
1950
1950
|
let fallback: number | null = null;
|
|
@@ -2649,16 +2649,16 @@ async function crossRegionVotes(
|
|
|
2649
2649
|
const consumed = new Set<number>();
|
|
2650
2650
|
let probes = 0;
|
|
2651
2651
|
// When atoms themselves are hubs (atomIsHub — a single byte reaches ≥ √N
|
|
2652
|
-
// contexts,
|
|
2653
|
-
// cross-region junction walks are dominated by the drift through
|
|
2654
|
-
// content's ancestry.
|
|
2655
|
-
// √N·W budget (profiled: 160,210 junction pops, 31% of think at
|
|
2656
|
-
//
|
|
2657
|
-
//
|
|
2652
|
+
// contexts, bounded-reads.md's own predicate), the corpus is large enough
|
|
2653
|
+
// that the cross-region junction walks are dominated by the drift through
|
|
2654
|
+
// common content's ancestry. Each of k candidate pairs otherwise spends its
|
|
2655
|
+
// own √N·W budget (profiled: 160,210 junction pops, 31% of think at N =
|
|
2656
|
+
// 325,608), and a cumulative dialogue multiplies bounded work into tens of
|
|
2657
|
+
// seconds. The structural walk is therefore given ONE k·W allowance per
|
|
2658
2658
|
// evidence tier, shared across every pair — k pairs × W phrase-scale levels,
|
|
2659
2659
|
// the minimal exact check; a pair whose container is not reached within it
|
|
2660
|
-
// falls through to the resonance tier (the ANN proposes what the shallow
|
|
2661
|
-
//
|
|
2660
|
+
// falls through to the resonance tier (the ANN proposes what the shallow walk
|
|
2661
|
+
// no longer exhaustively scans, exact-vs-approximate.md).
|
|
2662
2662
|
//
|
|
2663
2663
|
// Below atomIsHub the store is small and atoms still discriminate, so the
|
|
2664
2664
|
// walks keep exhaustive exact traversal (per-walk √N·W) — the shared budget
|
package/src/mind/bridge.ts
CHANGED
|
@@ -148,23 +148,23 @@ export function dismissedKnownContent(
|
|
|
148
148
|
return false;
|
|
149
149
|
}
|
|
150
150
|
|
|
151
|
-
// The seeded aligner this file used to own now lives in the shared match
|
|
152
|
-
//
|
|
153
|
-
//
|
|
154
|
-
//
|
|
151
|
+
// The seeded aligner this file used to own now lives in the shared match family
|
|
152
|
+
// as {@link alignAround} — the frame reading (match.ts) reads the same gaps and
|
|
153
|
+
// asks the OPPOSITE question of them (see AlignGap's own doc). Two consumers,
|
|
154
|
+
// one definition (factored-machinery.md); the bridge's reading is unchanged.
|
|
155
155
|
const align = alignAround;
|
|
156
156
|
|
|
157
157
|
/** Recall's corroborated-substitution bridge — see the module comment.
|
|
158
158
|
* Returns the best bridged grounding proposal, or null. */
|
|
159
159
|
/** `proposed` is a THUNK, not a list: the bridge's own cheap gates (the
|
|
160
|
-
* two-quantum query floor and the O(|query|) stored-window anchor scan)
|
|
161
|
-
*
|
|
162
|
-
*
|
|
163
|
-
*
|
|
164
|
-
*
|
|
165
|
-
*
|
|
166
|
-
*
|
|
167
|
-
*
|
|
160
|
+
* two-quantum query floor and the O(|query|) stored-window anchor scan) decide
|
|
161
|
+
* whether ANY candidate can be aligned, and they need no proposals to do it.
|
|
162
|
+
* Resolving the caller's proposals eagerly meant recall paid its exhaustive
|
|
163
|
+
* whole-index resonance — the most expensive single act on the refusal path —
|
|
164
|
+
* for every query, including the ones whose windows the store has never seen
|
|
165
|
+
* and which the anchor scan rejects outright. Same investment discipline the
|
|
166
|
+
* mechanism floors follow (mechanism-market.md): never compute a shared
|
|
167
|
+
* analysis just to discard it. */
|
|
168
168
|
export async function substitutionBridge(
|
|
169
169
|
ctx: MindContext,
|
|
170
170
|
query: Uint8Array,
|
|
@@ -290,12 +290,12 @@ async function bridgeImpl(
|
|
|
290
290
|
);
|
|
291
291
|
return null;
|
|
292
292
|
}
|
|
293
|
-
// NO DISCRIMINATING LITERAL EVIDENCE — abstain (
|
|
294
|
-
// through the literal spans it did NOT substitute; those anchors are
|
|
295
|
-
// whole of its evidence.
|
|
296
|
-
// clamped at the √N hub bound, i.e. the window is corpus-global
|
|
297
|
-
// — the query's unsubstituted part discriminates nothing, and the
|
|
298
|
-
// substituted span is carrying the entire semantic load.
|
|
293
|
+
// NO DISCRIMINATING LITERAL EVIDENCE — abstain (INVARIANTS.md). A bridge
|
|
294
|
+
// grounds through the literal spans it did NOT substitute; those anchors are
|
|
295
|
+
// the whole of its evidence. When every one of them is SATURATED —
|
|
296
|
+
// containment clamped at the √N hub bound, i.e. the window is corpus-global
|
|
297
|
+
// scaffolding — the query's unsubstituted part discriminates nothing, and the
|
|
298
|
+
// single substituted span is carrying the entire semantic load. That is not a
|
|
299
299
|
// corroborated bridge; it is a template match, and it FABRICATES.
|
|
300
300
|
//
|
|
301
301
|
// Measured on the trained store (hubBound 571). "What is the capital of"
|
|
@@ -310,7 +310,8 @@ async function bridgeImpl(
|
|
|
310
310
|
// them silent and cannot be credited for them.
|
|
311
311
|
//
|
|
312
312
|
// This introduces NO new threshold: `bound` is the same √N reading of "hub"
|
|
313
|
-
// the anchor scan already clamps its own containment read to (
|
|
313
|
+
// the anchor scan already clamps its own containment read to (thresholds.md,
|
|
314
|
+
// commonality.md).
|
|
314
315
|
if (allWindowsAreScaffolding(ctx, query)) {
|
|
315
316
|
ctx.trace?.step(
|
|
316
317
|
"substitutionBridge",
|
|
@@ -353,8 +354,8 @@ async function bridgeImpl(
|
|
|
353
354
|
//
|
|
354
355
|
// The question every gap poses is "may the two forms differ HERE without
|
|
355
356
|
// differing in what they SAY?", and that is the discriminative-vs-
|
|
356
|
-
// scaffolding question
|
|
357
|
-
// population.
|
|
357
|
+
// scaffolding question commonality.md names, over the CORPUS-GLOBAL
|
|
358
|
+
// population. It already has one definition — `dominates(reachOf(...), N)`,
|
|
358
359
|
// the same gate confluence's filler test uses ("scaffolding never binds").
|
|
359
360
|
// Nothing new is derived here; the bar is read, not invented.
|
|
360
361
|
//
|
|
@@ -366,17 +367,17 @@ async function bridgeImpl(
|
|
|
366
367
|
// climb's own definition of non-discriminative), or it resolves to a
|
|
367
368
|
// majority of the corpus's contexts. "the process of ", " is the ".
|
|
368
369
|
//
|
|
369
|
-
// THE READING MATTERS, not just the population
|
|
370
|
-
// deliberately does NOT go through `reachOf`, which maps BOTH "saturated"
|
|
371
|
-
//
|
|
372
|
-
//
|
|
373
|
-
//
|
|
374
|
-
//
|
|
375
|
-
//
|
|
376
|
-
//
|
|
377
|
-
//
|
|
378
|
-
//
|
|
379
|
-
//
|
|
370
|
+
// THE READING MATTERS, not just the population — see commonality.md. This
|
|
371
|
+
// deliberately does NOT go through `reachOf`, which maps BOTH "saturated" and
|
|
372
|
+
// "reaches nothing" to Infinity. For IDF weighting those are the same thing
|
|
373
|
+
// (no usable identity evidence); for THIS question they are opposites — a
|
|
374
|
+
// window reaching nothing is novel content, the most discriminative material
|
|
375
|
+
// there is, and reading it as Infinity would call it scaffolding. Measured:
|
|
376
|
+
// with `reachOf`, "Is water wet?" was answered with "No, heavy water is not
|
|
377
|
+
// wet." — "heav"/"eavy" occur once, reach no edge-bearing ancestor, and were
|
|
378
|
+
// written off as filler. So an empty-rooted window is NEVER explained, and
|
|
379
|
+
// neither is an untrained one (the same principle attestedQ applies to the
|
|
380
|
+
// query side).
|
|
380
381
|
const reachMemo = sharedReachMemo(ctx);
|
|
381
382
|
const explainedSpan = (
|
|
382
383
|
bytes: Uint8Array,
|
|
@@ -0,0 +1,202 @@
|
|
|
1
|
+
// corpus.ts — read the trained memory back out of the DAG, AS DATA.
|
|
2
|
+
//
|
|
3
|
+
// A trained experience pair IS one continuation edge: `src` is the context that
|
|
4
|
+
// was deposited, `dst` is what the mind learnt follows it. Reading them back
|
|
5
|
+
// uses the store's own structure and its own indexes — no auxiliary index is
|
|
6
|
+
// built, nothing is written, and NOTHING HERE KNOWS ABOUT TEXT: this layer takes
|
|
7
|
+
// bytes and returns bytes. The text case is one helper on the Mind
|
|
8
|
+
// (`searchCorpusText`), which encodes, calls this, and decodes.
|
|
9
|
+
//
|
|
10
|
+
// WHERE EACH STAGE COMES FROM (ported from the demo's `explore.ts`, which
|
|
11
|
+
// hand-rolled its own resolution):
|
|
12
|
+
//
|
|
13
|
+
// 1. PERCEIVE, CONTENT-ADDRESS and ADMIT the query — `recognise()`, the SAME
|
|
14
|
+
// machinery an answer goes through. It returns the sites: the query spans
|
|
15
|
+
// that content-addressed to a stored node that can lead somewhere. The
|
|
16
|
+
// demo's own recursive `findLeaf`/`findBranch` walk was a second
|
|
17
|
+
// implementation of exactly this, and it is NOT ported.
|
|
18
|
+
// 2. CLIMB the structural `kid` table from each resolved site to the
|
|
19
|
+
// edge-bearing contexts above it (`edgeAncestors`, traverse.ts), weighting
|
|
20
|
+
// each context by how much query content reached it.
|
|
21
|
+
// 3. READ the continuation off the edge table (`nextFirst`).
|
|
22
|
+
//
|
|
23
|
+
// COST is set by how much of the QUERY resolves, never by the size of the
|
|
24
|
+
// store: the sites are what recognition already found, the climb is bounded by
|
|
25
|
+
// the declared `corpusClimbs`/`corpusContextsPerClimb`, and every store call is
|
|
26
|
+
// a point probe or a capped read. All work is accounted by the store's own
|
|
27
|
+
// meter hooks when a response's meter is open — there is no second instrument.
|
|
28
|
+
//
|
|
29
|
+
// WHAT THIS IS NOT. Exact content addressing, not fuzzy keyword search: a query
|
|
30
|
+
// shares results with a stored note when it shares actual chunk-aligned content
|
|
31
|
+
// with it. An arbitrary mid-word fragment resolves to nothing, and the honest
|
|
32
|
+
// answer there is "nothing matched" — which is why `sampleCorpus` exists, and
|
|
33
|
+
// why the miss is reported as a STATE rather than as prose (the text helper
|
|
34
|
+
// turns it into words).
|
|
35
|
+
|
|
36
|
+
import { recognise } from "./recognition.js";
|
|
37
|
+
import { edgeAncestors } from "./traverse.js";
|
|
38
|
+
import type { MindContext } from "./types.js";
|
|
39
|
+
|
|
40
|
+
/** One stored experience pair, as bytes. */
|
|
41
|
+
export interface CorpusPair {
|
|
42
|
+
context: Uint8Array;
|
|
43
|
+
continuation: Uint8Array;
|
|
44
|
+
contextId: number;
|
|
45
|
+
continuationId: number;
|
|
46
|
+
/** Bytes of the query this pair was matched on — 0 when browsing. */
|
|
47
|
+
matchedBytes: number;
|
|
48
|
+
/** True when the stored bytes ran past the declared preview capacity. */
|
|
49
|
+
contextTruncated: boolean;
|
|
50
|
+
continuationTruncated: boolean;
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
/** Why a search produced no pairs. A STATE, so a caller's own layer can say it
|
|
54
|
+
* in its own words — the byte layer does not speak. */
|
|
55
|
+
export type CorpusMiss = "matched" | "nothing-resolved" | "no-continuations";
|
|
56
|
+
|
|
57
|
+
export interface CorpusResult {
|
|
58
|
+
pairs: CorpusPair[];
|
|
59
|
+
/** Query subtrees that content-addressed to a real stored node. */
|
|
60
|
+
resolved: number;
|
|
61
|
+
/** Distinct edge-bearing contexts the climb reached. */
|
|
62
|
+
reached: number;
|
|
63
|
+
/** Distinct contexts that carry a learnt continuation, store-wide. */
|
|
64
|
+
totalContexts: number;
|
|
65
|
+
/** True when these are browse samples rather than search results. */
|
|
66
|
+
browsed: boolean;
|
|
67
|
+
miss: CorpusMiss;
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/** A context node as a pair, or null when it carries no continuation. */
|
|
71
|
+
function pairOf(
|
|
72
|
+
ctx: MindContext,
|
|
73
|
+
id: number,
|
|
74
|
+
matchedBytes: number,
|
|
75
|
+
): CorpusPair | null {
|
|
76
|
+
const outs = ctx.store.nextFirst(id, 1);
|
|
77
|
+
if (outs.length === 0) return null;
|
|
78
|
+
const cap = ctx.cfg.corpusPreviewBytes;
|
|
79
|
+
const context = ctx.store.bytesPrefix(id, cap + 1);
|
|
80
|
+
const continuation = ctx.store.bytesPrefix(outs[0], cap + 1);
|
|
81
|
+
const contextTruncated = context.length > cap;
|
|
82
|
+
const continuationTruncated = continuation.length > cap;
|
|
83
|
+
return {
|
|
84
|
+
context: contextTruncated ? context.subarray(0, cap) : context,
|
|
85
|
+
continuation: continuationTruncated
|
|
86
|
+
? continuation.subarray(0, cap)
|
|
87
|
+
: continuation,
|
|
88
|
+
contextId: id,
|
|
89
|
+
continuationId: outs[0],
|
|
90
|
+
matchedBytes,
|
|
91
|
+
contextTruncated,
|
|
92
|
+
continuationTruncated,
|
|
93
|
+
};
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
/** Which stored notes does this query reach? BYTES in, BYTES out.
|
|
97
|
+
*
|
|
98
|
+
* Exact content addressing through the machinery that already exists: the
|
|
99
|
+
* query's recognised sites are the resolved subtrees, the climb goes up from
|
|
100
|
+
* the biggest first, and a pair is a context that carries a continuation. */
|
|
101
|
+
export function searchCorpus(
|
|
102
|
+
ctx: MindContext,
|
|
103
|
+
queryBytes: Uint8Array,
|
|
104
|
+
limit?: number,
|
|
105
|
+
): CorpusResult {
|
|
106
|
+
const store = ctx.store;
|
|
107
|
+
const want = Math.max(
|
|
108
|
+
1,
|
|
109
|
+
Math.min(limit ?? ctx.cfg.corpusContextsPerClimb, ctx.cfg.corpusLimitMax),
|
|
110
|
+
);
|
|
111
|
+
// A resolved subtree must account for at least one window (W): a single
|
|
112
|
+
// character resolves against almost any store and means nothing. W is the
|
|
113
|
+
// mind's own line between chance and evidence — derived, never declared.
|
|
114
|
+
const floor = ctx.space.maxGroup;
|
|
115
|
+
const resolved = recognise(ctx, queryBytes).sites
|
|
116
|
+
.map((s) => ({ id: s.payload, len: store.contentLen(s.payload, 512) }))
|
|
117
|
+
.filter((r) => r.len >= floor);
|
|
118
|
+
// Biggest first, lowest id breaking ties: a clause is evidence, a character is
|
|
119
|
+
// noise, and equal evidence must not be decided by iteration order.
|
|
120
|
+
const byLength = resolved
|
|
121
|
+
.sort((a, b) => b.len - a.len || a.id - b.id)
|
|
122
|
+
.slice(0, ctx.cfg.corpusClimbs);
|
|
123
|
+
|
|
124
|
+
// Weight each context by how much query content reached it.
|
|
125
|
+
const weight = new Map<number, number>();
|
|
126
|
+
for (const { id, len } of byLength) {
|
|
127
|
+
for (
|
|
128
|
+
const root of edgeAncestors(ctx, id, ctx.cfg.corpusContextsPerClimb).roots
|
|
129
|
+
) {
|
|
130
|
+
weight.set(root, (weight.get(root) ?? 0) + len);
|
|
131
|
+
}
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
const pairs: CorpusPair[] = [];
|
|
135
|
+
const ranked = [...weight.entries()].sort(
|
|
136
|
+
(a, b) => b[1] - a[1] || a[0] - b[0],
|
|
137
|
+
);
|
|
138
|
+
for (const [id, w] of ranked) {
|
|
139
|
+
if (pairs.length >= want) break;
|
|
140
|
+
const pair = pairOf(ctx, id, w);
|
|
141
|
+
if (pair) pairs.push(pair);
|
|
142
|
+
}
|
|
143
|
+
return {
|
|
144
|
+
pairs,
|
|
145
|
+
resolved: byLength.length,
|
|
146
|
+
reached: weight.size,
|
|
147
|
+
totalContexts: store.edgeSourceCount(),
|
|
148
|
+
browsed: false,
|
|
149
|
+
miss: pairs.length > 0
|
|
150
|
+
? "matched"
|
|
151
|
+
: weight.size === 0
|
|
152
|
+
? "nothing-resolved"
|
|
153
|
+
: "no-continuations",
|
|
154
|
+
};
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
/** Browse real pairs, striding the id space so the sample is spread rather than
|
|
158
|
+
* one local cluster. DETERMINISTIC: `from` is the caller's own offset, so
|
|
159
|
+
* browsing twice with different offsets shows different notes without a random
|
|
160
|
+
* draw (the demo drew `Math.random()`, which the engine cannot do — same seed,
|
|
161
|
+
* same order, same query must mean the same answer). */
|
|
162
|
+
export function sampleCorpus(
|
|
163
|
+
ctx: MindContext,
|
|
164
|
+
limit?: number,
|
|
165
|
+
from = 0,
|
|
166
|
+
): CorpusResult {
|
|
167
|
+
const store = ctx.store;
|
|
168
|
+
const want = Math.max(
|
|
169
|
+
1,
|
|
170
|
+
Math.min(limit ?? ctx.cfg.corpusContextsPerClimb, ctx.cfg.corpusLimitMax),
|
|
171
|
+
);
|
|
172
|
+
const total = store.nodeCount();
|
|
173
|
+
const probes = ctx.cfg.corpusSampleProbes;
|
|
174
|
+
const floorBytes = ctx.cfg.corpusSampleFloorBytes;
|
|
175
|
+
const pairs: CorpusPair[] = [];
|
|
176
|
+
// EACH CONTEXT AT MOST ONCE. Striding the id space revisits ids when the
|
|
177
|
+
// store is small relative to the probe budget (measured: a 160-node store
|
|
178
|
+
// returned the SAME pair six times for `limit: 6`), and a browse that repeats
|
|
179
|
+
// itself is not a browse. The demo had the same hole; it is invisible only on
|
|
180
|
+
// a store far larger than the probe budget.
|
|
181
|
+
const seen = new Set<number>();
|
|
182
|
+
for (let i = 0; i < probes && pairs.length < want && total > 0; i++) {
|
|
183
|
+
const slot = (i / probes + from) % 1;
|
|
184
|
+
const id = Math.floor(slot * total);
|
|
185
|
+
if (seen.has(id)) continue;
|
|
186
|
+
if (!store.has(id) || !store.hasNext(id)) continue;
|
|
187
|
+
if (store.contentLen(id, floorBytes) < floorBytes) continue;
|
|
188
|
+
const pair = pairOf(ctx, id, 0);
|
|
189
|
+
if (pair) {
|
|
190
|
+
seen.add(id);
|
|
191
|
+
pairs.push(pair);
|
|
192
|
+
}
|
|
193
|
+
}
|
|
194
|
+
return {
|
|
195
|
+
pairs,
|
|
196
|
+
resolved: 0,
|
|
197
|
+
reached: pairs.length,
|
|
198
|
+
totalContexts: store.edgeSourceCount(),
|
|
199
|
+
browsed: true,
|
|
200
|
+
miss: pairs.length > 0 ? "matched" : "no-continuations",
|
|
201
|
+
};
|
|
202
|
+
}
|