@hviana/sema 0.8.0 → 0.8.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +22 -1
- package/DATASETS.md +1 -1
- package/dist/example/train_base/config.js +2 -2
- package/dist/example/train_base/corpora/massive.js +1 -1
- package/dist/example/train_base/readers.js +1 -1
- package/dist/src/geometry.d.ts +10 -10
- package/dist/src/geometry.js +25 -24
- package/dist/src/meter.d.ts +4 -12
- package/dist/src/meter.js +14 -14
- package/dist/src/mind/attention.js +12 -12
- package/dist/src/mind/bridge.d.ts +8 -8
- package/dist/src/mind/bridge.js +33 -32
- package/dist/src/mind/graph-search.d.ts +0 -8
- package/dist/src/mind/graph-search.js +9 -8
- package/dist/src/mind/junction.d.ts +1 -1
- package/dist/src/mind/junction.js +8 -8
- package/dist/src/mind/learning.js +36 -35
- package/dist/src/mind/match.js +14 -13
- package/dist/src/mind/mechanisms/cover.js +13 -12
- package/dist/src/mind/mechanisms/prefix-completion.js +24 -24
- package/dist/src/mind/mechanisms/recall.js +38 -40
- package/dist/src/mind/mechanisms/reference.js +16 -16
- package/dist/src/mind/mind.d.ts +6 -7
- package/dist/src/mind/pipeline-mechanism.d.ts +10 -8
- package/dist/src/mind/pipeline-mechanism.js +25 -21
- package/dist/src/mind/pipeline.d.ts +9 -9
- package/dist/src/mind/pipeline.js +24 -23
- package/dist/src/mind/primitives.d.ts +5 -5
- package/dist/src/mind/primitives.js +5 -5
- package/dist/src/mind/recognition.d.ts +14 -13
- package/dist/src/mind/recognition.js +23 -23
- package/dist/src/mind/resonance.js +21 -21
- package/dist/src/mind/traverse.d.ts +54 -52
- package/dist/src/mind/traverse.js +74 -72
- package/dist/src/mind/types.d.ts +4 -4
- package/dist/src/store.d.ts +12 -12
- package/dist/src/store.js +12 -12
- package/docs/INDEX.md +2 -2
- package/docs/architecture/exact-vs-approximate.md +2 -1
- package/docs/architecture/fold-contract.md +1 -1
- package/docs/failures/tempting-but-wrong.md +2 -3
- package/docs/harness/gates.md +7 -7
- package/example/train_base/config.ts +2 -2
- package/example/train_base/corpora/massive.ts +1 -1
- package/example/train_base/readers.ts +1 -1
- package/jsr.json +1 -1
- package/package.json +1 -1
- package/src/geometry.ts +25 -24
- package/src/meter.ts +14 -14
- package/src/mind/attention.ts +12 -12
- package/src/mind/bridge.ts +33 -32
- package/src/mind/graph-search.ts +9 -8
- package/src/mind/junction.ts +8 -8
- package/src/mind/learning.ts +36 -35
- package/src/mind/match.ts +20 -19
- package/src/mind/mechanisms/cover.ts +13 -12
- package/src/mind/mechanisms/prefix-completion.ts +24 -24
- package/src/mind/mechanisms/recall.ts +38 -40
- package/src/mind/mechanisms/reference.ts +16 -16
- package/src/mind/mind.ts +6 -7
- package/src/mind/pipeline-mechanism.ts +25 -21
- package/src/mind/pipeline.ts +33 -32
- package/src/mind/primitives.ts +5 -5
- package/src/mind/recognition.ts +23 -23
- package/src/mind/resonance.ts +21 -21
- package/src/mind/traverse.ts +74 -72
- package/src/mind/types.ts +4 -4
- package/src/store.ts +20 -20
- package/test/08-storage.test.mjs +1 -1
- package/test/35-prefix-edge.test.mjs +1 -1
- package/test/40-choosenext-scale-guard.test.mjs +16 -17
- package/test/56-bridge-identity-admission.test.mjs +6 -6
- package/test/70-prefix-completion.test.mjs +4 -3
- package/test/72-prefix-candidate-supply.test.mjs +3 -3
- package/test/73-scaffolding-only-bridge-abstains.test.mjs +6 -6
- package/test/75-multiturn-context-optimisation.test.mjs +5 -5
- package/test/84-composed-answer-honesty.test.mjs +5 -6
- package/test/88-dependency-footprint.test.mjs +1 -1
- package/test/89-completion-recursion.test.mjs +17 -14
- package/test/90-connector-read-cap.test.mjs +10 -8
- package/test/93-regime-prediction.test.mjs +10 -10
- package/test/94-cross-region-budget.test.mjs +2 -2
- package/test/95-wide-resonance-removed.test.mjs +8 -7
- package/test/96-bytes-walk-termination.test.mjs +3 -3
|
@@ -9,11 +9,11 @@ import { cosine } from "../vec.js";
|
|
|
9
9
|
import { gistOf, read } from "./primitives.js";
|
|
10
10
|
import { canonicalWindows, leafIdPrefix, leafIdRun } from "./canonical.js";
|
|
11
11
|
//
|
|
12
|
-
// Budgeted on the same terms as the reach memo below (
|
|
13
|
-
//
|
|
14
|
-
//
|
|
15
|
-
//
|
|
16
|
-
//
|
|
12
|
+
// Budgeted on the same terms as the reach memo below (caches.md): these three
|
|
13
|
+
// maps are cleared on every write, but a long read-only session over a large
|
|
14
|
+
// store converges on one entry per node per map with nothing to bound it. Past
|
|
15
|
+
// the cap all three are dropped together and re-derived, costing cold
|
|
16
|
+
// structural probes and never a wrong answer.
|
|
17
17
|
const STRUCT_MEMO_MAX = 100_000;
|
|
18
18
|
const structCaches = new WeakMap();
|
|
19
19
|
// ── The shared ancestor-reach memo ──────────────────────────────────────
|
|
@@ -31,20 +31,21 @@ const structCaches = new WeakMap();
|
|
|
31
31
|
// battery repeatedly reaches the same corpus scaffolding even when its
|
|
32
32
|
// surface questions differ.
|
|
33
33
|
//
|
|
34
|
-
// Budgeted, not unbounded (
|
|
35
|
-
//
|
|
34
|
+
// Budgeted, not unbounded (caches.md): past the cap the whole map is dropped
|
|
35
|
+
// and
|
|
36
|
+
// re-derived, costing a cold climb and never a wrong answer.
|
|
36
37
|
const REACH_MEMO_MAX = 100_000;
|
|
37
38
|
const reachCaches = new WeakMap();
|
|
38
39
|
/** The reach memo this ask should use — see the note above.
|
|
39
40
|
*
|
|
40
|
-
* A TRACED response always gets a fresh, empty one.
|
|
41
|
-
*
|
|
42
|
-
*
|
|
43
|
-
*
|
|
44
|
-
*
|
|
45
|
-
*
|
|
46
|
-
*
|
|
47
|
-
*
|
|
41
|
+
* A TRACED response always gets a fresh, empty one. `AncestorReach`'s
|
|
42
|
+
* `visited`/`maxDepth`/`saturation` fields are populated only when a trace is
|
|
43
|
+
* attached, so an entry deposited by an untraced earlier turn would silently
|
|
44
|
+
* black out the reach detail of a later traced one; and the trace's reach
|
|
45
|
+
* payload is serialised by ITERATING this map, which must therefore hold what
|
|
46
|
+
* THIS climb consulted, not the whole conversation's history. Consistent with
|
|
47
|
+
* memoization.md: a traced response is a different machine — never benchmark
|
|
48
|
+
* with a trace attached. */
|
|
48
49
|
export function sharedReachMemo(ctx) {
|
|
49
50
|
if (ctx.trace !== null || ctx.climbMemo === null)
|
|
50
51
|
return new Map();
|
|
@@ -426,13 +427,14 @@ export function bearsEdge(ctx, id) {
|
|
|
426
427
|
return cachedHasNext(ctx, id, getStructCache(ctx));
|
|
427
428
|
}
|
|
428
429
|
/** Whether a node LEADS SOMEWHERE — it bears a continuation edge or a halo.
|
|
429
|
-
*
|
|
430
|
-
*
|
|
431
|
-
*
|
|
432
|
-
*
|
|
433
|
-
*
|
|
434
|
-
*
|
|
435
|
-
*
|
|
430
|
+
* The admission predicate recognition filters sites with (cover.md): a form
|
|
431
|
+
* that
|
|
432
|
+
* leads nowhere contributes nothing to any derivation. Runs once per candidate
|
|
433
|
+
* span on the recognition hot path — `hasNext` is cached per response (the same
|
|
434
|
+
* flat-branch ids are probed across prefix variants by canonicalChunkId).
|
|
435
|
+
* `hasHalo` is not cached: it's a single indexed point probe per candidate, and
|
|
436
|
+
* the candidates that reach this check have already been filtered by hasNext
|
|
437
|
+
* above in edgeAncestors. */
|
|
436
438
|
export function leadsSomewhere(ctx, id) {
|
|
437
439
|
const memo = getStructCache(ctx);
|
|
438
440
|
if (cachedHasNext(ctx, id, memo))
|
|
@@ -476,10 +478,10 @@ function boundFor(contextCount) {
|
|
|
476
478
|
return Math.ceil(Math.sqrt(Math.max(2, contextCount)));
|
|
477
479
|
}
|
|
478
480
|
/** Cap a candidate list at the hub bound √N (insertion order) — the ONE
|
|
479
|
-
*
|
|
480
|
-
*
|
|
481
|
-
*
|
|
482
|
-
*
|
|
481
|
+
* fan-out convention every walk and disambiguation uses (see bounded-reads.md).
|
|
482
|
+
* A node connected to more than √N others is a hub whose individual connections
|
|
483
|
+
* carry ~no discriminative information; materialising or scoring them all would
|
|
484
|
+
* make single decisions scale with the corpus. */
|
|
483
485
|
export function hubCap(ctx, ids) {
|
|
484
486
|
const bound = hubBound(ctx);
|
|
485
487
|
return ids.length > bound ? ids.slice(0, bound) : ids;
|
|
@@ -512,16 +514,16 @@ export function contains(ctx, ancestor, descendant) {
|
|
|
512
514
|
* the EXACT half's veto on calling them synonyms.
|
|
513
515
|
*
|
|
514
516
|
* Halos measure company, and the strongest company any two forms can keep is
|
|
515
|
-
*
|
|
516
|
-
*
|
|
517
|
-
*
|
|
518
|
-
*
|
|
519
|
-
*
|
|
520
|
-
*
|
|
521
|
-
*
|
|
522
|
-
*
|
|
523
|
-
*
|
|
524
|
-
*
|
|
517
|
+
* standing next to each other: a question and its answer co-occur in every
|
|
518
|
+
* episode that taught the pair, so their halos SHOULD be similar, and on a
|
|
519
|
+
* conversational store they are (measured on the CONV fixture: consecutive
|
|
520
|
+
* turns at 0.809 against a 0.516 concept threshold). A gate reading halo cosine
|
|
521
|
+
* alone therefore reads adjacency as synonymy and revoices an answer in the
|
|
522
|
+
* words of the question it answers — "it hangs in madrid" spliced back into
|
|
523
|
+
* "where is it kept now". The distributional layer cannot tell the two
|
|
524
|
+
* relations apart, because to it they are the same observation; the exact half
|
|
525
|
+
* can, for free, because it stored the edge. halo-sketch.md's division of
|
|
526
|
+
* labour exactly: approximate proposes, exact decides.
|
|
525
527
|
*
|
|
526
528
|
* Read LIMITed in both directions at the hub bound — a common continuation's
|
|
527
529
|
* fan-in is corpus-sized, and no single decision may scale with it. */
|
|
@@ -645,19 +647,18 @@ export function chooseNext(ctx, id, guide) {
|
|
|
645
647
|
// NO consensusFloor gate here (tried and reverted — see
|
|
646
648
|
// test/40-choosenext-scale-guard.test.mjs): that floor is calibrated for
|
|
647
649
|
// POOLED, IDF-weighted CLIMB VOTES (recallByResonance, commitVotes), where
|
|
648
|
-
// each corroborating region contributes at most ln N and the floor grows
|
|
649
|
-
//
|
|
650
|
-
//
|
|
651
|
-
//
|
|
652
|
-
//
|
|
653
|
-
// N-
|
|
654
|
-
//
|
|
655
|
-
//
|
|
656
|
-
//
|
|
657
|
-
//
|
|
658
|
-
//
|
|
659
|
-
//
|
|
660
|
-
// chooseNext pseudocode, which has no such floor.
|
|
650
|
+
// each corroborating region contributes at most ln N and the floor grows with
|
|
651
|
+
// N exactly as that per-region ceiling does (thresholds.md). `bestSupport`
|
|
652
|
+
// here is a different kind of quantity — a raw prevCount of how many training
|
|
653
|
+
// contexts predicted ONE destination, bounded by how often that specific fact
|
|
654
|
+
// was retold, never by corpus size N. Gating an N-invariant count against an
|
|
655
|
+
// N-growing threshold guarantees failure once N is large enough, discarding
|
|
656
|
+
// genuinely, structurally dominant edges (observed: a fact corroborated
|
|
657
|
+
// 2-to-1-1-1 refused at N≈325K, falling back to a noisy concept-hop). The
|
|
658
|
+
// loop above already IS the "genuinely competing" test: a tie leaves
|
|
659
|
+
// first-inserted as the pick (test/30's own pinned behaviour); a strict
|
|
660
|
+
// winner is real evidence regardless of corpus scale. Matches `chooseNext`'s
|
|
661
|
+
// own pseudocode, which has no such floor.
|
|
661
662
|
// Trace is built lazily — the filter + map below only execute when a
|
|
662
663
|
// trace listener is attached, so the common (no-trace) path pays only
|
|
663
664
|
// for the prevCount calls in the loop above, never for extra rItemShort
|
|
@@ -709,12 +710,13 @@ function rItemShort(ctx, id, role, score) {
|
|
|
709
710
|
* W-window it spells is contained by more places than the hub bound allows,
|
|
710
711
|
* i.e. the whole query is corpus-global scaffolding.
|
|
711
712
|
*
|
|
712
|
-
*
|
|
713
|
-
*
|
|
714
|
-
*
|
|
715
|
-
*
|
|
716
|
-
*
|
|
717
|
-
*
|
|
713
|
+
* WHAT IT IS FOR. Several mechanisms ground a query through the literal spans
|
|
714
|
+
* it
|
|
715
|
+
* did NOT explain, and those spans are the whole of their evidence. When every
|
|
716
|
+
* one of them is a hub, the query says nothing the corpus can be held to, and
|
|
717
|
+
* grounding it means picking one of thousands of continuations it gives no
|
|
718
|
+
* evidence for — a fabrication whatever the answer happens to be. Answering
|
|
719
|
+
* with silence there is the honest degradation contract (INVARIANTS.md).
|
|
718
720
|
*
|
|
719
721
|
* MEASURED SEPARATION (trained store, hubBound 571) — this is categorical,
|
|
720
722
|
* not marginal, and it is why the predicate lives here rather than being
|
|
@@ -731,11 +733,11 @@ function rItemShort(ctx, id, role, score) {
|
|
|
731
733
|
* evidence and sit on the SAME side as the correct ones, so this predicate
|
|
732
734
|
* is not what makes them silent and cannot be credited for them.
|
|
733
735
|
*
|
|
734
|
-
* NO NEW THRESHOLD (
|
|
735
|
-
*
|
|
736
|
-
*
|
|
737
|
-
*
|
|
738
|
-
*
|
|
736
|
+
* NO NEW THRESHOLD (thresholds.md): `hubBound` is the √N reading of "hub" used
|
|
737
|
+
* everywhere, and the containment read is clamped to it exactly as every other
|
|
738
|
+
* fan-out read is (bounded-reads.md). A query with no stored window at all is
|
|
739
|
+
* NOT scaffolding-only — it has no evidence either way, and its callers already
|
|
740
|
+
* refuse it on their own terms. */
|
|
739
741
|
export function allWindowsAreScaffolding(ctx, query) {
|
|
740
742
|
const W = ctx.space.maxGroup;
|
|
741
743
|
const bound = hubBound(ctx);
|
|
@@ -786,19 +788,19 @@ export function allWindowsAreScaffolding(ctx, query) {
|
|
|
786
788
|
* by climbing containment then parents. Nothing is added to the write side;
|
|
787
789
|
* this reads an index training already built.
|
|
788
790
|
*
|
|
789
|
-
* BOUNDED (
|
|
790
|
-
*
|
|
791
|
-
*
|
|
792
|
-
*
|
|
793
|
-
*
|
|
794
|
-
*
|
|
795
|
-
*
|
|
796
|
-
*
|
|
797
|
-
*
|
|
791
|
+
* BOUNDED (bounded-reads.md), AND WITH NO NEW THRESHOLD. The window whose
|
|
792
|
+
* containment is SMALLEST carries the most evidence, and one saturated at
|
|
793
|
+
* `hubBound` carries none — that is the same √N reading of "hub" the rest of
|
|
794
|
+
* the mind uses, not a tuned knob. The upward walk spends a budget of
|
|
795
|
+
* `hubBound` nodes and fans out by W, so a hub query enumerates nothing and the
|
|
796
|
+
* caller stays silent rather than guessing (INVARIANTS.md). Measured on the
|
|
797
|
+
* trained store: the photosynthesis form at a one-byte truncation picks a
|
|
798
|
+
* window with 52 containers, visits 446 nodes, and yields exactly ONE candidate
|
|
799
|
+
* that survives the caller's byte compare — the form itself.
|
|
798
800
|
*
|
|
799
|
-
* These are PROPOSALS only.
|
|
800
|
-
*
|
|
801
|
-
*
|
|
801
|
+
* These are PROPOSALS only. Every candidate still faces the byte-exact prefix
|
|
802
|
+
* compare and all three guards below, so a wrong proposal costs one bounded
|
|
803
|
+
* read and can never be voiced (exact-vs-approximate.md). */
|
|
802
804
|
export function formsOpenedBy(ctx, query) {
|
|
803
805
|
const store = ctx.store;
|
|
804
806
|
const W = ctx.space.maxGroup;
|
package/dist/src/mind/types.d.ts
CHANGED
|
@@ -250,10 +250,10 @@ export type AItem = {
|
|
|
250
250
|
export interface MindContext extends GraphSearchHost {
|
|
251
251
|
store: Store;
|
|
252
252
|
/** The work accumulator for the inference call in flight, or null when
|
|
253
|
-
*
|
|
254
|
-
*
|
|
255
|
-
*
|
|
256
|
-
*
|
|
253
|
+
* nothing is profiling — see src/meter.ts. WRITE-ONLY from the engine's point
|
|
254
|
+
* of view: no inference decision may read a counter, or determinism is gone
|
|
255
|
+
* (determinism.md). Every call site is `ctx.meter?.x++`, so an unprofiled
|
|
256
|
+
* response allocates nothing. */
|
|
257
257
|
meter: Meter | null;
|
|
258
258
|
space: Space;
|
|
259
259
|
alphabet: Alphabet;
|
package/dist/src/store.d.ts
CHANGED
|
@@ -701,18 +701,18 @@ export declare abstract class AbstractStore implements Store {
|
|
|
701
701
|
* remainders must fit the budget. Scattered differences leave a wide
|
|
702
702
|
* middle and are rejected.
|
|
703
703
|
*
|
|
704
|
-
* Every read here is CAPPED (
|
|
705
|
-
*
|
|
706
|
-
*
|
|
707
|
-
*
|
|
708
|
-
*
|
|
709
|
-
*
|
|
710
|
-
*
|
|
711
|
-
*
|
|
712
|
-
*
|
|
713
|
-
*
|
|
714
|
-
*
|
|
715
|
-
*
|
|
704
|
+
* Every read here is CAPPED (bounded-reads.md). It used to open with
|
|
705
|
+
* `bytesPrefix(k, Number.MAX_SAFE_INTEGER)` — the ALL sentinel, i.e. the full
|
|
706
|
+
* materialising `bytes()` read — on the deposit hot path, and only then
|
|
707
|
+
* compare lengths. So a candidate the length test was about to reject had
|
|
708
|
+
* already been reconstructed byte for byte. The LENGTHS decide first instead,
|
|
709
|
+
* from the `contentLen` memo the interning order has already built bottom-up,
|
|
710
|
+
* and the target's length is itself read under a cap: a target longer than
|
|
711
|
+
* `la + W` is rejected without touching one of its bytes. Same semantics —
|
|
712
|
+
* the old capped `b` read would have produced `a.length + W + 1` here and
|
|
713
|
+
* failed the very same test — strictly fewer byte reads. The `+ 1` on each
|
|
714
|
+
* byte cap keeps `_prefix`'s "complete reconstruction" test true, so the
|
|
715
|
+
* results still cache. */
|
|
716
716
|
private differsByOneWindow;
|
|
717
717
|
putLeaf(bytes: Uint8Array, gist: Vec): Promise<NodeId>;
|
|
718
718
|
putBranch(kids: NodeId[], gist: Vec): Promise<NodeId>;
|
package/dist/src/store.js
CHANGED
|
@@ -1235,18 +1235,18 @@ export class AbstractStore {
|
|
|
1235
1235
|
* remainders must fit the budget. Scattered differences leave a wide
|
|
1236
1236
|
* middle and are rejected.
|
|
1237
1237
|
*
|
|
1238
|
-
* Every read here is CAPPED (
|
|
1239
|
-
*
|
|
1240
|
-
*
|
|
1241
|
-
*
|
|
1242
|
-
*
|
|
1243
|
-
*
|
|
1244
|
-
*
|
|
1245
|
-
*
|
|
1246
|
-
*
|
|
1247
|
-
*
|
|
1248
|
-
*
|
|
1249
|
-
*
|
|
1238
|
+
* Every read here is CAPPED (bounded-reads.md). It used to open with
|
|
1239
|
+
* `bytesPrefix(k, Number.MAX_SAFE_INTEGER)` — the ALL sentinel, i.e. the full
|
|
1240
|
+
* materialising `bytes()` read — on the deposit hot path, and only then
|
|
1241
|
+
* compare lengths. So a candidate the length test was about to reject had
|
|
1242
|
+
* already been reconstructed byte for byte. The LENGTHS decide first instead,
|
|
1243
|
+
* from the `contentLen` memo the interning order has already built bottom-up,
|
|
1244
|
+
* and the target's length is itself read under a cap: a target longer than
|
|
1245
|
+
* `la + W` is rejected without touching one of its bytes. Same semantics —
|
|
1246
|
+
* the old capped `b` read would have produced `a.length + W + 1` here and
|
|
1247
|
+
* failed the very same test — strictly fewer byte reads. The `+ 1` on each
|
|
1248
|
+
* byte cap keeps `_prefix`'s "complete reconstruction" test true, so the
|
|
1249
|
+
* results still cache. */
|
|
1250
1250
|
differsByOneWindow(kids, targetId, W) {
|
|
1251
1251
|
const lens = kids.map((k) => this.contentLen(k));
|
|
1252
1252
|
let la = 0;
|
package/docs/INDEX.md
CHANGED
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
# Sema Documentation Index
|
|
2
2
|
|
|
3
3
|
Sema is a single system stated three ways: the law lives in `docs/architecture/`
|
|
4
|
-
(what holds), the prescription in `AGENTS.md`
|
|
5
|
-
|
|
4
|
+
(what holds), the prescription in `AGENTS.md` (what to do and where), and the
|
|
5
|
+
proof in `test/` (pins that fail when the law is broken).
|
|
6
6
|
|
|
7
7
|
## Routing — what to read for each task
|
|
8
8
|
|
|
@@ -44,4 +44,5 @@ Two rules in `attention.ts` encode "exact decides" and must not be flattened:
|
|
|
44
44
|
|
|
45
45
|
Add a tier to the shared family in `mind/match.ts` with a derived gate
|
|
46
46
|
(`geometry.ts`), never a private `score >= k` check. A new mechanism is a
|
|
47
|
-
`(matcher, direction, gate)` configuration over that family
|
|
47
|
+
`(matcher, direction, gate)` configuration over that family
|
|
48
|
+
(`docs/architecture/match-project.md`).
|
|
@@ -76,7 +76,7 @@ not to do, why it fails, and what to do instead.
|
|
|
76
76
|
- **WHY:** `frameSlots` reports (contracted gaps tagged
|
|
77
77
|
substitution/insertion/deletion); `carriesFillers` judges;
|
|
78
78
|
`Precomputed.frames` inventories — elects nothing (`AGENTS §2` Cross-cutting
|
|
79
|
-
contracts — `docs/architecture/
|
|
79
|
+
contracts — `docs/architecture/match-project.md` § Frame reading).
|
|
80
80
|
- **CORRECT:** Report everything in the shared layer; apply
|
|
81
81
|
`substituteAll(contA, fillersA→fillersB)==contB` and
|
|
82
82
|
frame-dominance/`W`-reach/distinctness in the consumer (reference). Pinned by
|
|
@@ -88,8 +88,7 @@ not to do, why it fails, and what to do instead.
|
|
|
88
88
|
`depth[i]`/`dominates(depth, aligned)` to decide climb/IDF.
|
|
89
89
|
- **WHY:** They measure different things: global reach (minority discriminates,
|
|
90
90
|
powers climb/pooling) vs weave-local depth with `MIN_WEAVE=2` (what the local
|
|
91
|
-
cohort shares, powers CAST) (`
|
|
92
|
-
measures of commonality).
|
|
91
|
+
cohort shares, powers CAST) (`docs/architecture/commonality.md`).
|
|
93
92
|
- **CORRECT:** Climb/attention uses corpus-global;
|
|
94
93
|
`frame(i) ⇔ depth[i]>MIN_WEAVE ∧ dominates(depth[i],aligned)` for CAST. Pinned
|
|
95
94
|
by `test/50-cast-analog-consensus-floor.test.mjs` and
|
package/docs/harness/gates.md
CHANGED
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
Four executable gates. Each: run the command, check what it guards, follow its
|
|
4
4
|
§.
|
|
5
5
|
|
|
6
|
-
## 1 — Correctness (all
|
|
6
|
+
## 1 — Correctness (all 90 suites)
|
|
7
7
|
|
|
8
8
|
```bash
|
|
9
9
|
npm test
|
|
@@ -13,8 +13,8 @@ Guards honest silence, determinism, and every pinned contract. Silence:
|
|
|
13
13
|
unrelated queries ground to nothing (`test/28`, `50`, `56`, `67`, `76`, `84`).
|
|
14
14
|
Determinism: same seed + deposit order + query gives byte-identical answer
|
|
15
15
|
(`test/20`). Every invariant is pinned — a simplification that fails a test is
|
|
16
|
-
wrong until the test is shown wrong. §14–25 (pipeline), §
|
|
17
|
-
AGENTS.md §2 invariants 1–5.
|
|
16
|
+
wrong until the test is shown wrong. §14–25 (pipeline), §64 (derived
|
|
17
|
+
thresholds), AGENTS.md §2 invariants 1–5.
|
|
18
18
|
|
|
19
19
|
## 2 — Work accounting (profiler)
|
|
20
20
|
|
|
@@ -27,8 +27,8 @@ Guards without trace: counters deterministic and diffable between runs; phases
|
|
|
27
27
|
nest (not disjoint — `think` contains every mechanism phase); shared analyses
|
|
28
28
|
charged to themselves, not to the first toucher; millisecond fields are
|
|
29
29
|
non-deterministic hints only. With `--trace`, recognition idempotence still
|
|
30
|
-
holds (`test/42`). `src/meter.ts`, `docs/architecture/meter.md`, §
|
|
31
|
-
§
|
|
30
|
+
holds (`test/42`). `src/meter.ts`, `docs/architecture/meter.md`, §55,
|
|
31
|
+
`AGENTS.md` §6.
|
|
32
32
|
|
|
33
33
|
## 3 — Dependency footprint
|
|
34
34
|
|
|
@@ -38,7 +38,7 @@ node --test test/88-dependency-footprint.test.mjs
|
|
|
38
38
|
|
|
39
39
|
Guards `dist/src` imports only `node:` + relative paths, and `package.json`
|
|
40
40
|
declares no `dependencies` (examples use `devDependencies` lazily). The
|
|
41
|
-
near-zero footprint is a product feature. AGENTS.md §
|
|
41
|
+
near-zero footprint is a product feature. AGENTS.md §7, §3 (store has one
|
|
42
42
|
runtime dep: `node:sqlite`).
|
|
43
43
|
|
|
44
44
|
## 4 — Fold invariance and sublinear scaling
|
|
@@ -53,4 +53,4 @@ positional; grid regression (14.3% survival) cannot pass. `14` — inference cos
|
|
|
53
53
|
is sublinear in corpus size (power-law exponent ≪ 1) and constant-rate in input
|
|
54
54
|
length; measured on independent disjoint corpora via log–log slope.
|
|
55
55
|
`src/geometry.ts` (`contentLevels`), `docs/architecture/fold-contract.md` +
|
|
56
|
-
`bounded-reads.md`, §
|
|
56
|
+
`bounded-reads.md`, §59, §63.
|
|
@@ -4,8 +4,8 @@
|
|
|
4
4
|
// the cache ceiling, the read budgets, the caps. A knob that describes ONE
|
|
5
5
|
// CORPUS (which pairs of SmolSent, how many SODA dialogues, how long an Aya
|
|
6
6
|
// field may be) belongs next to that corpus's adapter, together with the
|
|
7
|
-
// evidence that fixed its default —
|
|
8
|
-
// constraint
|
|
7
|
+
// evidence that fixed its default — a comment carries the constraint, and a
|
|
8
|
+
// constraint is only readable beside the code it constrains.
|
|
9
9
|
|
|
10
10
|
import { join } from "node:path";
|
|
11
11
|
|
|
@@ -37,7 +37,7 @@ import { convertedParquetUnits } from "./converted-parquet.js";
|
|
|
37
37
|
//
|
|
38
38
|
// So it displaces some wrong answers and manufactures others, INCLUDING turning
|
|
39
39
|
// a correct silence into a wrong answer — and honest silence is a stated
|
|
40
|
-
// property of this engine (
|
|
40
|
+
// property of this engine (INVARIANTS.md). On the mixed-curriculum store the
|
|
41
41
|
// same shape produced the fragment "nus" for "wake me up at nine am".
|
|
42
42
|
//
|
|
43
43
|
// That evidence is four probes on toy stores and is NOT conclusive; it is,
|
|
@@ -13,7 +13,7 @@
|
|
|
13
13
|
//
|
|
14
14
|
// THE ONLY THIRD-PARTY CODE IN THIS REPOSITORY IS BELOW, and it is LAZILY
|
|
15
15
|
// LOADED. Sema itself imports nothing outside `node:` — that is a product
|
|
16
|
-
// property, not an accident (AGENTS.md §
|
|
16
|
+
// property, not an accident (AGENTS.md §7) — and this trainer is an EXAMPLE,
|
|
17
17
|
// not part of the library. hyparquet (+ its Snappy codec) is therefore a dev
|
|
18
18
|
// dependency, and it is loaded by a dynamic import the first time a Parquet
|
|
19
19
|
// corpus is actually read: a curriculum with no Parquet stage (SmolSent,
|
package/jsr.json
CHANGED
package/package.json
CHANGED
package/src/geometry.ts
CHANGED
|
@@ -349,19 +349,20 @@ function bytesToLeaves(
|
|
|
349
349
|
* sentences fall would be importing an assumption the architecture rejects.
|
|
350
350
|
* Random binary must, and does, behave exactly like prose.
|
|
351
351
|
*
|
|
352
|
-
* Every constant is derived (
|
|
353
|
-
*
|
|
354
|
-
*
|
|
355
|
-
*
|
|
356
|
-
*
|
|
357
|
-
*
|
|
358
|
-
*
|
|
359
|
-
*
|
|
360
|
-
*
|
|
361
|
-
*
|
|
362
|
-
*
|
|
363
|
-
*
|
|
364
|
-
*
|
|
352
|
+
* Every constant is derived (thresholds.md): the cut mask is W, so a cut is
|
|
353
|
+
* offered once per quantum of bytes — which, composed with the minimum below,
|
|
354
|
+
* puts the expected segment at minLen + W − 1 ≈ 6 B rather than at W,
|
|
355
|
+
* deliberately (see the refutation recorded at `cutRate` in {@link
|
|
356
|
+
* contentLevels}: a segment is the flat PHRASE-scale unit the W-ary groups are
|
|
357
|
+
* built from, not a group of W children, and forcing E[len] = W costs 15
|
|
358
|
+
* tests). The minimum is W−1, `canonicalWindows`'s straddle neighbour and the
|
|
359
|
+
* write side's own floor for a unit; and the maximum is the KEYRING's seat
|
|
360
|
+
* count, because a segment folds as ONE flat node and `fold` has exactly that
|
|
361
|
+
* many seats to bind children into. Capping there is what keeps the fold light:
|
|
362
|
+
* a segment of 3..seats leaves is a single node, where splitting it into
|
|
363
|
+
* W-groups plus a remainder would cost two or three and the remainders barely
|
|
364
|
+
* share (measured: partial-arity nodes 504 → 3,590, and total distinct nodes
|
|
365
|
+
* 8,142 → 9,712, when segments folded as [W][rest]). */
|
|
365
366
|
/** {@link contentBoundaries} plus, for each cut, its LEVEL — how deep in the
|
|
366
367
|
* tree that cut reaches.
|
|
367
368
|
*
|
|
@@ -622,16 +623,16 @@ export function knownPrefixLength(
|
|
|
622
623
|
* correct boundary. Pass them through from `perceive`; the geometry
|
|
623
624
|
* computes the stable prefix internally.
|
|
624
625
|
*
|
|
625
|
-
* `boundaries` is the CALLER-computed stable-prefix boundary set
|
|
626
|
-
*
|
|
627
|
-
*
|
|
628
|
-
*
|
|
629
|
-
*
|
|
630
|
-
*
|
|
631
|
-
*
|
|
632
|
-
*
|
|
633
|
-
*
|
|
634
|
-
*
|
|
626
|
+
* `boundaries` is the CALLER-computed stable-prefix boundary set
|
|
627
|
+
* (fold-contract.md): strictly-increasing proper byte offsets, each the length
|
|
628
|
+
* of a prefix that is already a stored whole-stream form. When given, the fold
|
|
629
|
+
* splits into the segments between consecutive boundaries — each folded
|
|
630
|
+
* independently, exactly as it folded when it was learned — and the segment
|
|
631
|
+
* roots join LEFT-NESTED (((s₀·s₁)·s₂)…), so every learnt cumulative-context
|
|
632
|
+
* root reappears as an identical subtree (and, by hash-consing, the very same
|
|
633
|
+
* node) inside the grown stream. This is what lets a conversation's next turn
|
|
634
|
+
* extend perception instead of refolding it: identical prefixes produce
|
|
635
|
+
* identical subtrees regardless of what follows them. */
|
|
635
636
|
export function bytesToTree(
|
|
636
637
|
space: Space,
|
|
637
638
|
alphabet: Alphabet,
|
|
@@ -962,7 +963,7 @@ function flatFold(
|
|
|
962
963
|
return { tree: sema(gist, null, kids), len: n };
|
|
963
964
|
}
|
|
964
965
|
|
|
965
|
-
|
|
966
|
+
/* * The stable-prefix segmented fold (fold-contract.md). Each segment between
|
|
966
967
|
* consecutive boundaries folds PLAINLY and independently; segment roots
|
|
967
968
|
* join left-nested, and only the final root is normalized (the linear-fold
|
|
968
969
|
* contract: one normalize per perception). A segment's own inner splits
|
package/src/meter.ts
CHANGED
|
@@ -8,8 +8,8 @@
|
|
|
8
8
|
// Four contracts, all load-bearing:
|
|
9
9
|
//
|
|
10
10
|
// 1. NEVER READ BY INFERENCE. No counter may reach a decision, a threshold,
|
|
11
|
-
// or an ordering. Determinism (
|
|
12
|
-
// meter is write-only from the engine's point of view.
|
|
11
|
+
// or an ordering. Determinism (determinism.md) survives
|
|
12
|
+
// only because the meter is write-only from the engine's point of view.
|
|
13
13
|
// 2. OFF BY DEFAULT, AND FREE WHEN OFF. Every call site is `meter?.x++` on
|
|
14
14
|
// a null field. Nothing allocates, nothing is keyed, nothing is timed
|
|
15
15
|
// unless a Meter is attached (`new Mind({ profile: true })`).
|
|
@@ -79,9 +79,9 @@ export class Meter {
|
|
|
79
79
|
nodeRecords = 0;
|
|
80
80
|
/** `store.bytes` / `store.bytesPrefix` — one reconstruction request. */
|
|
81
81
|
byteReads = 0;
|
|
82
|
-
|
|
83
|
-
*
|
|
84
|
-
*
|
|
82
|
+
/* * Bytes actually handed back by those reads — the real I/O volume, and the
|
|
83
|
+
* number that exposes an unbounded read (bounded-reads.md) that a call count
|
|
84
|
+
* alone hides. */
|
|
85
85
|
bytesRead = 0;
|
|
86
86
|
/** `store.contentLen`. */
|
|
87
87
|
lenReads = 0;
|
|
@@ -178,10 +178,10 @@ export class Meter {
|
|
|
178
178
|
junctionPops = 0;
|
|
179
179
|
/** Ascents that ended by EXHAUSTING the expansion budget rather than by
|
|
180
180
|
* deciding — the walk abstained and the caller silently fell through to a
|
|
181
|
-
*
|
|
182
|
-
*
|
|
183
|
-
*
|
|
184
|
-
*
|
|
181
|
+
* lower tier of the ladder — honest degradation, and nothing else reports it
|
|
182
|
+
* (INVARIANTS.md). It rises the moment a SHARED budget is drained by an
|
|
183
|
+
* earlier walk, which is what makes "this tier answered nothing"
|
|
184
|
+
* distinguishable from "this tier never got to look". */
|
|
185
185
|
junctionBudgetExhausted = 0;
|
|
186
186
|
/** Arbitrary byte spans whose distributional company was VSA-bundled from
|
|
187
187
|
* existing episode halos. */
|
|
@@ -238,11 +238,11 @@ export class Meter {
|
|
|
238
238
|
}
|
|
239
239
|
}
|
|
240
240
|
|
|
241
|
-
|
|
242
|
-
*
|
|
243
|
-
*
|
|
244
|
-
*
|
|
245
|
-
*
|
|
241
|
+
/* * Time one SYNCHRONOUS phase. The sync/async seam is a real contract
|
|
242
|
+
* (meter.md) — perception, recognition and the graph search are synchronous —
|
|
243
|
+
* so a synchronous layer must not be wrapped in `time`'s promise just to be
|
|
244
|
+
* measured: that would make the profiled path await where the unprofiled one
|
|
245
|
+
* does not, and a meter never changes what a layer computes. */
|
|
246
246
|
timeSync<T>(phase: string, fn: () => T): T {
|
|
247
247
|
const before = this.snapshot();
|
|
248
248
|
const t = performance.now();
|
package/src/mind/attention.ts
CHANGED
|
@@ -1941,10 +1941,10 @@ export function canonicalChunkId(
|
|
|
1941
1941
|
// CAST lost a point of attention it needed (test/29 D1/D2).
|
|
1942
1942
|
//
|
|
1943
1943
|
// So scan every offset and prefer an anchor that still discriminates: not
|
|
1944
|
-
// saturated, and among those the one reaching the FEWEST contexts
|
|
1945
|
-
// corpus-global).
|
|
1946
|
-
// old generalising choice stand — there is then no
|
|
1947
|
-
// find, and abstaining is the honest outcome.
|
|
1944
|
+
// saturated, and among those the one reaching the FEWEST contexts
|
|
1945
|
+
// (commonality.md, corpus-global). Only when every window in the region
|
|
1946
|
+
// saturates does the old generalising choice stand — there is then no
|
|
1947
|
+
// discriminative anchor to find, and abstaining is the honest outcome.
|
|
1948
1948
|
let discId: number | null = null;
|
|
1949
1949
|
let discReached = Infinity;
|
|
1950
1950
|
let fallback: number | null = null;
|
|
@@ -2649,16 +2649,16 @@ async function crossRegionVotes(
|
|
|
2649
2649
|
const consumed = new Set<number>();
|
|
2650
2650
|
let probes = 0;
|
|
2651
2651
|
// When atoms themselves are hubs (atomIsHub — a single byte reaches ≥ √N
|
|
2652
|
-
// contexts,
|
|
2653
|
-
// cross-region junction walks are dominated by the drift through
|
|
2654
|
-
// content's ancestry.
|
|
2655
|
-
// √N·W budget (profiled: 160,210 junction pops, 31% of think at
|
|
2656
|
-
//
|
|
2657
|
-
//
|
|
2652
|
+
// contexts, bounded-reads.md's own predicate), the corpus is large enough
|
|
2653
|
+
// that the cross-region junction walks are dominated by the drift through
|
|
2654
|
+
// common content's ancestry. Each of k candidate pairs otherwise spends its
|
|
2655
|
+
// own √N·W budget (profiled: 160,210 junction pops, 31% of think at N =
|
|
2656
|
+
// 325,608), and a cumulative dialogue multiplies bounded work into tens of
|
|
2657
|
+
// seconds. The structural walk is therefore given ONE k·W allowance per
|
|
2658
2658
|
// evidence tier, shared across every pair — k pairs × W phrase-scale levels,
|
|
2659
2659
|
// the minimal exact check; a pair whose container is not reached within it
|
|
2660
|
-
// falls through to the resonance tier (the ANN proposes what the shallow
|
|
2661
|
-
//
|
|
2660
|
+
// falls through to the resonance tier (the ANN proposes what the shallow walk
|
|
2661
|
+
// no longer exhaustively scans, exact-vs-approximate.md).
|
|
2662
2662
|
//
|
|
2663
2663
|
// Below atomIsHub the store is small and atoms still discriminate, so the
|
|
2664
2664
|
// walks keep exhaustive exact traversal (per-walk √N·W) — the shared budget
|