@hviana/sema 0.8.0 → 0.8.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +22 -1
- package/DATASETS.md +1 -1
- package/dist/example/train_base/config.js +2 -2
- package/dist/example/train_base/corpora/massive.js +1 -1
- package/dist/example/train_base/readers.js +1 -1
- package/dist/src/config.d.ts +17 -0
- package/dist/src/config.js +18 -0
- package/dist/src/geometry.d.ts +10 -10
- package/dist/src/geometry.js +25 -24
- package/dist/src/meter.d.ts +29 -12
- package/dist/src/meter.js +58 -14
- package/dist/src/mind/attention.js +12 -12
- package/dist/src/mind/bridge.d.ts +8 -8
- package/dist/src/mind/bridge.js +33 -32
- package/dist/src/mind/corpus.d.ts +40 -0
- package/dist/src/mind/corpus.js +149 -0
- package/dist/src/mind/graph-search.d.ts +7 -8
- package/dist/src/mind/graph-search.js +244 -32
- package/dist/src/mind/index.d.ts +3 -1
- package/dist/src/mind/index.js +1 -0
- package/dist/src/mind/junction.d.ts +1 -1
- package/dist/src/mind/junction.js +8 -8
- package/dist/src/mind/learning.js +36 -35
- package/dist/src/mind/match.d.ts +8 -3
- package/dist/src/mind/match.js +156 -71
- package/dist/src/mind/mechanisms/cast.js +18 -2
- package/dist/src/mind/mechanisms/cover.js +19 -12
- package/dist/src/mind/mechanisms/prefix-completion.js +24 -24
- package/dist/src/mind/mechanisms/recall.js +38 -40
- package/dist/src/mind/mechanisms/reference.js +16 -16
- package/dist/src/mind/mind.d.ts +61 -7
- package/dist/src/mind/mind.js +72 -2
- package/dist/src/mind/pipeline-mechanism.d.ts +10 -8
- package/dist/src/mind/pipeline-mechanism.js +25 -21
- package/dist/src/mind/pipeline.d.ts +9 -9
- package/dist/src/mind/pipeline.js +49 -29
- package/dist/src/mind/primitives.d.ts +5 -5
- package/dist/src/mind/primitives.js +5 -5
- package/dist/src/mind/reasoning.d.ts +5 -1
- package/dist/src/mind/reasoning.js +54 -1
- package/dist/src/mind/recognition.d.ts +14 -13
- package/dist/src/mind/recognition.js +23 -23
- package/dist/src/mind/resonance.js +21 -21
- package/dist/src/mind/traverse.d.ts +54 -52
- package/dist/src/mind/traverse.js +83 -73
- package/dist/src/mind/types.d.ts +26 -4
- package/dist/src/store.d.ts +12 -12
- package/dist/src/store.js +12 -12
- package/docs/INDEX.md +2 -2
- package/docs/architecture/exact-vs-approximate.md +2 -1
- package/docs/architecture/fold-contract.md +1 -1
- package/docs/failures/tempting-but-wrong.md +33 -5
- package/docs/harness/gates.md +7 -7
- package/example/train_base/config.ts +2 -2
- package/example/train_base/corpora/massive.ts +1 -1
- package/example/train_base/readers.ts +1 -1
- package/jsr.json +1 -1
- package/package.json +1 -1
- package/src/config.ts +35 -0
- package/src/geometry.ts +25 -24
- package/src/meter.ts +61 -14
- package/src/mind/attention.ts +12 -12
- package/src/mind/bridge.ts +33 -32
- package/src/mind/corpus.ts +202 -0
- package/src/mind/graph-search.ts +261 -31
- package/src/mind/index.ts +8 -1
- package/src/mind/junction.ts +8 -8
- package/src/mind/learning.ts +36 -35
- package/src/mind/match.ts +163 -73
- package/src/mind/mechanisms/cast.ts +17 -1
- package/src/mind/mechanisms/cover.ts +18 -12
- package/src/mind/mechanisms/prefix-completion.ts +24 -24
- package/src/mind/mechanisms/recall.ts +38 -40
- package/src/mind/mechanisms/reference.ts +16 -16
- package/src/mind/mind.ts +129 -7
- package/src/mind/pipeline-mechanism.ts +25 -21
- package/src/mind/pipeline.ts +63 -38
- package/src/mind/primitives.ts +5 -5
- package/src/mind/reasoning.ts +55 -0
- package/src/mind/recognition.ts +23 -23
- package/src/mind/resonance.ts +21 -21
- package/src/mind/traverse.ts +83 -73
- package/src/mind/types.ts +30 -4
- package/src/store.ts +20 -20
- package/test/08-storage.test.mjs +1 -1
- package/test/100-complete-grounding-trace.test.mjs +109 -0
- package/test/101-alignment-gap-bound.test.mjs +106 -0
- package/test/102-production-composes-at-scale.test.mjs +110 -0
- package/test/103-alignment-gap-budget.test.mjs +89 -0
- package/test/104-composition-is-reported.test.mjs +90 -0
- package/test/105-derive-through-reports-its-refusal.test.mjs +113 -0
- package/test/106-the-join-fires.test.mjs +94 -0
- package/test/107-the-join-is-counted.test.mjs +81 -0
- package/test/108-the-join-chains.test.mjs +78 -0
- package/test/109-the-pivot-is-counted.test.mjs +60 -0
- package/test/110-the-reasoner-stops-when-the-question-is-answered.test.mjs +91 -0
- package/test/111-the-cover-assembly-is-counted.test.mjs +74 -0
- package/test/112-the-exploration-does-not-grow-with-the-hub.test.mjs +89 -0
- package/test/113-the-rationale-payload-is-bounded.test.mjs +84 -0
- package/test/114-alignment-budget-is-per-sweep.test.mjs +93 -0
- package/test/116-the-extension-is-gated-by-the-pipelines-own-remainder.test.mjs +100 -0
- package/test/117-corpus-search.test.mjs +171 -0
- package/test/14-scaling.test.mjs +10 -7
- package/test/35-prefix-edge.test.mjs +1 -1
- package/test/40-choosenext-scale-guard.test.mjs +16 -17
- package/test/56-bridge-identity-admission.test.mjs +6 -6
- package/test/70-prefix-completion.test.mjs +4 -3
- package/test/72-prefix-candidate-supply.test.mjs +3 -3
- package/test/73-scaffolding-only-bridge-abstains.test.mjs +6 -6
- package/test/75-multiturn-context-optimisation.test.mjs +5 -5
- package/test/76-reference-binding.test.mjs +6 -1
- package/test/84-composed-answer-honesty.test.mjs +5 -6
- package/test/88-dependency-footprint.test.mjs +1 -1
- package/test/89-completion-recursion.test.mjs +47 -19
- package/test/90-connector-read-cap.test.mjs +10 -8
- package/test/93-regime-prediction.test.mjs +10 -10
- package/test/94-cross-region-budget.test.mjs +2 -2
- package/test/95-wide-resonance-removed.test.mjs +8 -7
- package/test/96-bytes-walk-termination.test.mjs +3 -3
|
@@ -9,11 +9,11 @@ import { cosine } from "../vec.js";
|
|
|
9
9
|
import { gistOf, read } from "./primitives.js";
|
|
10
10
|
import { canonicalWindows, leafIdPrefix, leafIdRun } from "./canonical.js";
|
|
11
11
|
//
|
|
12
|
-
// Budgeted on the same terms as the reach memo below (
|
|
13
|
-
//
|
|
14
|
-
//
|
|
15
|
-
//
|
|
16
|
-
//
|
|
12
|
+
// Budgeted on the same terms as the reach memo below (caches.md): these three
|
|
13
|
+
// maps are cleared on every write, but a long read-only session over a large
|
|
14
|
+
// store converges on one entry per node per map with nothing to bound it. Past
|
|
15
|
+
// the cap all three are dropped together and re-derived, costing cold
|
|
16
|
+
// structural probes and never a wrong answer.
|
|
17
17
|
const STRUCT_MEMO_MAX = 100_000;
|
|
18
18
|
const structCaches = new WeakMap();
|
|
19
19
|
// ── The shared ancestor-reach memo ──────────────────────────────────────
|
|
@@ -31,20 +31,21 @@ const structCaches = new WeakMap();
|
|
|
31
31
|
// battery repeatedly reaches the same corpus scaffolding even when its
|
|
32
32
|
// surface questions differ.
|
|
33
33
|
//
|
|
34
|
-
// Budgeted, not unbounded (
|
|
35
|
-
//
|
|
34
|
+
// Budgeted, not unbounded (caches.md): past the cap the whole map is dropped
|
|
35
|
+
// and
|
|
36
|
+
// re-derived, costing a cold climb and never a wrong answer.
|
|
36
37
|
const REACH_MEMO_MAX = 100_000;
|
|
37
38
|
const reachCaches = new WeakMap();
|
|
38
39
|
/** The reach memo this ask should use — see the note above.
|
|
39
40
|
*
|
|
40
|
-
* A TRACED response always gets a fresh, empty one.
|
|
41
|
-
*
|
|
42
|
-
*
|
|
43
|
-
*
|
|
44
|
-
*
|
|
45
|
-
*
|
|
46
|
-
*
|
|
47
|
-
*
|
|
41
|
+
* A TRACED response always gets a fresh, empty one. `AncestorReach`'s
|
|
42
|
+
* `visited`/`maxDepth`/`saturation` fields are populated only when a trace is
|
|
43
|
+
* attached, so an entry deposited by an untraced earlier turn would silently
|
|
44
|
+
* black out the reach detail of a later traced one; and the trace's reach
|
|
45
|
+
* payload is serialised by ITERATING this map, which must therefore hold what
|
|
46
|
+
* THIS climb consulted, not the whole conversation's history. Consistent with
|
|
47
|
+
* memoization.md: a traced response is a different machine — never benchmark
|
|
48
|
+
* with a trace attached. */
|
|
48
49
|
export function sharedReachMemo(ctx) {
|
|
49
50
|
if (ctx.trace !== null || ctx.climbMemo === null)
|
|
50
51
|
return new Map();
|
|
@@ -426,13 +427,14 @@ export function bearsEdge(ctx, id) {
|
|
|
426
427
|
return cachedHasNext(ctx, id, getStructCache(ctx));
|
|
427
428
|
}
|
|
428
429
|
/** Whether a node LEADS SOMEWHERE — it bears a continuation edge or a halo.
|
|
429
|
-
*
|
|
430
|
-
*
|
|
431
|
-
*
|
|
432
|
-
*
|
|
433
|
-
*
|
|
434
|
-
*
|
|
435
|
-
*
|
|
430
|
+
* The admission predicate recognition filters sites with (cover.md): a form
|
|
431
|
+
* that
|
|
432
|
+
* leads nowhere contributes nothing to any derivation. Runs once per candidate
|
|
433
|
+
* span on the recognition hot path — `hasNext` is cached per response (the same
|
|
434
|
+
* flat-branch ids are probed across prefix variants by canonicalChunkId).
|
|
435
|
+
* `hasHalo` is not cached: it's a single indexed point probe per candidate, and
|
|
436
|
+
* the candidates that reach this check have already been filtered by hasNext
|
|
437
|
+
* above in edgeAncestors. */
|
|
436
438
|
export function leadsSomewhere(ctx, id) {
|
|
437
439
|
const memo = getStructCache(ctx);
|
|
438
440
|
if (cachedHasNext(ctx, id, memo))
|
|
@@ -476,10 +478,10 @@ function boundFor(contextCount) {
|
|
|
476
478
|
return Math.ceil(Math.sqrt(Math.max(2, contextCount)));
|
|
477
479
|
}
|
|
478
480
|
/** Cap a candidate list at the hub bound √N (insertion order) — the ONE
|
|
479
|
-
*
|
|
480
|
-
*
|
|
481
|
-
*
|
|
482
|
-
*
|
|
481
|
+
* fan-out convention every walk and disambiguation uses (see bounded-reads.md).
|
|
482
|
+
* A node connected to more than √N others is a hub whose individual connections
|
|
483
|
+
* carry ~no discriminative information; materialising or scoring them all would
|
|
484
|
+
* make single decisions scale with the corpus. */
|
|
483
485
|
export function hubCap(ctx, ids) {
|
|
484
486
|
const bound = hubBound(ctx);
|
|
485
487
|
return ids.length > bound ? ids.slice(0, bound) : ids;
|
|
@@ -512,16 +514,16 @@ export function contains(ctx, ancestor, descendant) {
|
|
|
512
514
|
* the EXACT half's veto on calling them synonyms.
|
|
513
515
|
*
|
|
514
516
|
* Halos measure company, and the strongest company any two forms can keep is
|
|
515
|
-
*
|
|
516
|
-
*
|
|
517
|
-
*
|
|
518
|
-
*
|
|
519
|
-
*
|
|
520
|
-
*
|
|
521
|
-
*
|
|
522
|
-
*
|
|
523
|
-
*
|
|
524
|
-
*
|
|
517
|
+
* standing next to each other: a question and its answer co-occur in every
|
|
518
|
+
* episode that taught the pair, so their halos SHOULD be similar, and on a
|
|
519
|
+
* conversational store they are (measured on the CONV fixture: consecutive
|
|
520
|
+
* turns at 0.809 against a 0.516 concept threshold). A gate reading halo cosine
|
|
521
|
+
* alone therefore reads adjacency as synonymy and revoices an answer in the
|
|
522
|
+
* words of the question it answers — "it hangs in madrid" spliced back into
|
|
523
|
+
* "where is it kept now". The distributional layer cannot tell the two
|
|
524
|
+
* relations apart, because to it they are the same observation; the exact half
|
|
525
|
+
* can, for free, because it stored the edge. halo-sketch.md's division of
|
|
526
|
+
* labour exactly: approximate proposes, exact decides.
|
|
525
527
|
*
|
|
526
528
|
* Read LIMITed in both directions at the hub bound — a common continuation's
|
|
527
529
|
* fan-in is corpus-sized, and no single decision may scale with it. */
|
|
@@ -645,25 +647,32 @@ export function chooseNext(ctx, id, guide) {
|
|
|
645
647
|
// NO consensusFloor gate here (tried and reverted — see
|
|
646
648
|
// test/40-choosenext-scale-guard.test.mjs): that floor is calibrated for
|
|
647
649
|
// POOLED, IDF-weighted CLIMB VOTES (recallByResonance, commitVotes), where
|
|
648
|
-
// each corroborating region contributes at most ln N and the floor grows
|
|
649
|
-
//
|
|
650
|
-
//
|
|
651
|
-
//
|
|
652
|
-
//
|
|
653
|
-
// N-
|
|
654
|
-
//
|
|
655
|
-
//
|
|
656
|
-
//
|
|
657
|
-
//
|
|
658
|
-
//
|
|
659
|
-
//
|
|
660
|
-
// chooseNext pseudocode, which has no such floor.
|
|
650
|
+
// each corroborating region contributes at most ln N and the floor grows with
|
|
651
|
+
// N exactly as that per-region ceiling does (thresholds.md). `bestSupport`
|
|
652
|
+
// here is a different kind of quantity — a raw prevCount of how many training
|
|
653
|
+
// contexts predicted ONE destination, bounded by how often that specific fact
|
|
654
|
+
// was retold, never by corpus size N. Gating an N-invariant count against an
|
|
655
|
+
// N-growing threshold guarantees failure once N is large enough, discarding
|
|
656
|
+
// genuinely, structurally dominant edges (observed: a fact corroborated
|
|
657
|
+
// 2-to-1-1-1 refused at N≈325K, falling back to a noisy concept-hop). The
|
|
658
|
+
// loop above already IS the "genuinely competing" test: a tie leaves
|
|
659
|
+
// first-inserted as the pick (test/30's own pinned behaviour); a strict
|
|
660
|
+
// winner is real evidence regardless of corpus scale. Matches `chooseNext`'s
|
|
661
|
+
// own pseudocode, which has no such floor.
|
|
661
662
|
// Trace is built lazily — the filter + map below only execute when a
|
|
662
663
|
// trace listener is attached, so the common (no-trace) path pays only
|
|
663
664
|
// for the prevCount calls in the loop above, never for extra rItemShort
|
|
664
665
|
// byte-reads.
|
|
665
666
|
if (ctx.trace) {
|
|
666
|
-
|
|
667
|
+
// A BOUNDED SAMPLE, AND THE COUNT. The step used to carry EVERY candidate
|
|
668
|
+
// it weighed — measured on the trained store, 1559 out-items in one step
|
|
669
|
+
// (hubBound's own size) and 1082 in another (the hub's degree). The
|
|
670
|
+
// rationale's job is to explain the CHOICE, and the count is what says how
|
|
671
|
+
// wide the field was; the declared candidate budget (`recallQueryK`) is what
|
|
672
|
+
// bounds the sample, so no number is invented here.
|
|
673
|
+
const others = capped
|
|
674
|
+
.filter((c) => c !== best)
|
|
675
|
+
.slice(0, ctx.cfg.rationaleSampleK);
|
|
667
676
|
ctx.trace.step("disambiguate", [rItemShort(ctx, best, "halo-evidence", bestSupport)], others.map((c) => rItemShort(ctx, c, "candidate", ctx.store.prevCount(c))), `${capped.length} continuations — distributional evidence selects ` +
|
|
668
677
|
`the most corroborated (distinct contexts ${bestSupport}, ` +
|
|
669
678
|
`poured mass ${bestMass})`);
|
|
@@ -709,12 +718,13 @@ function rItemShort(ctx, id, role, score) {
|
|
|
709
718
|
* W-window it spells is contained by more places than the hub bound allows,
|
|
710
719
|
* i.e. the whole query is corpus-global scaffolding.
|
|
711
720
|
*
|
|
712
|
-
*
|
|
713
|
-
*
|
|
714
|
-
*
|
|
715
|
-
*
|
|
716
|
-
*
|
|
717
|
-
*
|
|
721
|
+
* WHAT IT IS FOR. Several mechanisms ground a query through the literal spans
|
|
722
|
+
* it
|
|
723
|
+
* did NOT explain, and those spans are the whole of their evidence. When every
|
|
724
|
+
* one of them is a hub, the query says nothing the corpus can be held to, and
|
|
725
|
+
* grounding it means picking one of thousands of continuations it gives no
|
|
726
|
+
* evidence for — a fabrication whatever the answer happens to be. Answering
|
|
727
|
+
* with silence there is the honest degradation contract (INVARIANTS.md).
|
|
718
728
|
*
|
|
719
729
|
* MEASURED SEPARATION (trained store, hubBound 571) — this is categorical,
|
|
720
730
|
* not marginal, and it is why the predicate lives here rather than being
|
|
@@ -731,11 +741,11 @@ function rItemShort(ctx, id, role, score) {
|
|
|
731
741
|
* evidence and sit on the SAME side as the correct ones, so this predicate
|
|
732
742
|
* is not what makes them silent and cannot be credited for them.
|
|
733
743
|
*
|
|
734
|
-
* NO NEW THRESHOLD (
|
|
735
|
-
*
|
|
736
|
-
*
|
|
737
|
-
*
|
|
738
|
-
*
|
|
744
|
+
* NO NEW THRESHOLD (thresholds.md): `hubBound` is the √N reading of "hub" used
|
|
745
|
+
* everywhere, and the containment read is clamped to it exactly as every other
|
|
746
|
+
* fan-out read is (bounded-reads.md). A query with no stored window at all is
|
|
747
|
+
* NOT scaffolding-only — it has no evidence either way, and its callers already
|
|
748
|
+
* refuse it on their own terms. */
|
|
739
749
|
export function allWindowsAreScaffolding(ctx, query) {
|
|
740
750
|
const W = ctx.space.maxGroup;
|
|
741
751
|
const bound = hubBound(ctx);
|
|
@@ -786,19 +796,19 @@ export function allWindowsAreScaffolding(ctx, query) {
|
|
|
786
796
|
* by climbing containment then parents. Nothing is added to the write side;
|
|
787
797
|
* this reads an index training already built.
|
|
788
798
|
*
|
|
789
|
-
* BOUNDED (
|
|
790
|
-
*
|
|
791
|
-
*
|
|
792
|
-
*
|
|
793
|
-
*
|
|
794
|
-
*
|
|
795
|
-
*
|
|
796
|
-
*
|
|
797
|
-
*
|
|
799
|
+
* BOUNDED (bounded-reads.md), AND WITH NO NEW THRESHOLD. The window whose
|
|
800
|
+
* containment is SMALLEST carries the most evidence, and one saturated at
|
|
801
|
+
* `hubBound` carries none — that is the same √N reading of "hub" the rest of
|
|
802
|
+
* the mind uses, not a tuned knob. The upward walk spends a budget of
|
|
803
|
+
* `hubBound` nodes and fans out by W, so a hub query enumerates nothing and the
|
|
804
|
+
* caller stays silent rather than guessing (INVARIANTS.md). Measured on the
|
|
805
|
+
* trained store: the photosynthesis form at a one-byte truncation picks a
|
|
806
|
+
* window with 52 containers, visits 446 nodes, and yields exactly ONE candidate
|
|
807
|
+
* that survives the caller's byte compare — the form itself.
|
|
798
808
|
*
|
|
799
|
-
* These are PROPOSALS only.
|
|
800
|
-
*
|
|
801
|
-
*
|
|
809
|
+
* These are PROPOSALS only. Every candidate still faces the byte-exact prefix
|
|
810
|
+
* compare and all three guards below, so a wrong proposal costs one bounded
|
|
811
|
+
* read and can never be voiced (exact-vs-approximate.md). */
|
|
802
812
|
export function formsOpenedBy(ctx, query) {
|
|
803
813
|
const store = ctx.store;
|
|
804
814
|
const W = ctx.space.maxGroup;
|
package/dist/src/mind/types.d.ts
CHANGED
|
@@ -35,12 +35,34 @@ export interface GraphSearchHost {
|
|
|
35
35
|
starts: ReadonlySet<number>;
|
|
36
36
|
};
|
|
37
37
|
chooseNext?(node: number): number | undefined;
|
|
38
|
+
/** The boundary positions of `bytes` under the engine's ONE boundary rule
|
|
39
|
+
* (geometry.ts's `contentBoundaries`), or undefined when the host has no
|
|
40
|
+
* space to ask. The join's key is an entity plus a prefix of the tail, and
|
|
41
|
+
* the prefix that names a stored relation ENDS on one of these boundaries —
|
|
42
|
+
* measured, 5 of 5 accepted keys over four join-firing queries, where the
|
|
43
|
+
* byte-by-byte scan spent 153 probes for 14 boundaries. Boundaries are
|
|
44
|
+
* content-defined and STABLE under prefix extension, which is why a corpus
|
|
45
|
+
* key's end is a boundary of the query's own fold of the same bytes. */
|
|
46
|
+
contentCuts?(bytes: Uint8Array): readonly number[];
|
|
38
47
|
/** The admission predicate — `traverse.ts`'s `leadsSomewhere`, its ONE
|
|
39
48
|
* definition: does this node bear an edge or a halo? Optional, so a bare
|
|
40
49
|
* host (a raw Store and nothing else) still works; when present, the search
|
|
41
50
|
* uses it rather than re-probing the store, which keeps the predicate
|
|
42
51
|
* single-defined AND memoised on the response-scoped struct cache. */
|
|
43
52
|
leadsSomewhere?(id: number): boolean;
|
|
53
|
+
/** Report a SEARCH REFUSAL into the rationale — the channel AGENTS §6
|
|
54
|
+
* requires: a callback threaded through a call chain must FEED the
|
|
55
|
+
* rationale, the way `GraphSearch`'s `onDerivation` feeds `traceDerivation`,
|
|
56
|
+
* never a channel of its own. Optional, so a bare host stays silent rather
|
|
57
|
+
* than crashing. */
|
|
58
|
+
reportSearch?(name: string, parts: ReadonlyArray<Uint8Array>, note: string): void;
|
|
59
|
+
/** The CANONICAL resolver ({@link canonResolve}), optional like
|
|
60
|
+
* {@link leadsSomewhere}. The store's keys were written through the
|
|
61
|
+
* canonical fold, so a fact's `Gustaf Molander` and the deposited
|
|
62
|
+
* `gustaf molander` are the SAME node (measured inside a response: the
|
|
63
|
+
* canonical resolver maps the surface form to the deposited node while a raw
|
|
64
|
+
* resolve returns null). A bare host falls back to the plain probe. */
|
|
65
|
+
canonResolve?(bytes: Uint8Array): number | null;
|
|
44
66
|
}
|
|
45
67
|
export interface Recognition {
|
|
46
68
|
/** Forms that can lead somewhere — they have an edge or a halo. */
|
|
@@ -250,10 +272,10 @@ export type AItem = {
|
|
|
250
272
|
export interface MindContext extends GraphSearchHost {
|
|
251
273
|
store: Store;
|
|
252
274
|
/** The work accumulator for the inference call in flight, or null when
|
|
253
|
-
*
|
|
254
|
-
*
|
|
255
|
-
*
|
|
256
|
-
*
|
|
275
|
+
* nothing is profiling — see src/meter.ts. WRITE-ONLY from the engine's point
|
|
276
|
+
* of view: no inference decision may read a counter, or determinism is gone
|
|
277
|
+
* (determinism.md). Every call site is `ctx.meter?.x++`, so an unprofiled
|
|
278
|
+
* response allocates nothing. */
|
|
257
279
|
meter: Meter | null;
|
|
258
280
|
space: Space;
|
|
259
281
|
alphabet: Alphabet;
|
package/dist/src/store.d.ts
CHANGED
|
@@ -701,18 +701,18 @@ export declare abstract class AbstractStore implements Store {
|
|
|
701
701
|
* remainders must fit the budget. Scattered differences leave a wide
|
|
702
702
|
* middle and are rejected.
|
|
703
703
|
*
|
|
704
|
-
* Every read here is CAPPED (
|
|
705
|
-
*
|
|
706
|
-
*
|
|
707
|
-
*
|
|
708
|
-
*
|
|
709
|
-
*
|
|
710
|
-
*
|
|
711
|
-
*
|
|
712
|
-
*
|
|
713
|
-
*
|
|
714
|
-
*
|
|
715
|
-
*
|
|
704
|
+
* Every read here is CAPPED (bounded-reads.md). It used to open with
|
|
705
|
+
* `bytesPrefix(k, Number.MAX_SAFE_INTEGER)` — the ALL sentinel, i.e. the full
|
|
706
|
+
* materialising `bytes()` read — on the deposit hot path, and only then
|
|
707
|
+
* compare lengths. So a candidate the length test was about to reject had
|
|
708
|
+
* already been reconstructed byte for byte. The LENGTHS decide first instead,
|
|
709
|
+
* from the `contentLen` memo the interning order has already built bottom-up,
|
|
710
|
+
* and the target's length is itself read under a cap: a target longer than
|
|
711
|
+
* `la + W` is rejected without touching one of its bytes. Same semantics —
|
|
712
|
+
* the old capped `b` read would have produced `a.length + W + 1` here and
|
|
713
|
+
* failed the very same test — strictly fewer byte reads. The `+ 1` on each
|
|
714
|
+
* byte cap keeps `_prefix`'s "complete reconstruction" test true, so the
|
|
715
|
+
* results still cache. */
|
|
716
716
|
private differsByOneWindow;
|
|
717
717
|
putLeaf(bytes: Uint8Array, gist: Vec): Promise<NodeId>;
|
|
718
718
|
putBranch(kids: NodeId[], gist: Vec): Promise<NodeId>;
|
package/dist/src/store.js
CHANGED
|
@@ -1235,18 +1235,18 @@ export class AbstractStore {
|
|
|
1235
1235
|
* remainders must fit the budget. Scattered differences leave a wide
|
|
1236
1236
|
* middle and are rejected.
|
|
1237
1237
|
*
|
|
1238
|
-
* Every read here is CAPPED (
|
|
1239
|
-
*
|
|
1240
|
-
*
|
|
1241
|
-
*
|
|
1242
|
-
*
|
|
1243
|
-
*
|
|
1244
|
-
*
|
|
1245
|
-
*
|
|
1246
|
-
*
|
|
1247
|
-
*
|
|
1248
|
-
*
|
|
1249
|
-
*
|
|
1238
|
+
* Every read here is CAPPED (bounded-reads.md). It used to open with
|
|
1239
|
+
* `bytesPrefix(k, Number.MAX_SAFE_INTEGER)` — the ALL sentinel, i.e. the full
|
|
1240
|
+
* materialising `bytes()` read — on the deposit hot path, and only then
|
|
1241
|
+
* compare lengths. So a candidate the length test was about to reject had
|
|
1242
|
+
* already been reconstructed byte for byte. The LENGTHS decide first instead,
|
|
1243
|
+
* from the `contentLen` memo the interning order has already built bottom-up,
|
|
1244
|
+
* and the target's length is itself read under a cap: a target longer than
|
|
1245
|
+
* `la + W` is rejected without touching one of its bytes. Same semantics —
|
|
1246
|
+
* the old capped `b` read would have produced `a.length + W + 1` here and
|
|
1247
|
+
* failed the very same test — strictly fewer byte reads. The `+ 1` on each
|
|
1248
|
+
* byte cap keeps `_prefix`'s "complete reconstruction" test true, so the
|
|
1249
|
+
* results still cache. */
|
|
1250
1250
|
differsByOneWindow(kids, targetId, W) {
|
|
1251
1251
|
const lens = kids.map((k) => this.contentLen(k));
|
|
1252
1252
|
let la = 0;
|
package/docs/INDEX.md
CHANGED
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
# Sema Documentation Index
|
|
2
2
|
|
|
3
3
|
Sema is a single system stated three ways: the law lives in `docs/architecture/`
|
|
4
|
-
(what holds), the prescription in `AGENTS.md`
|
|
5
|
-
|
|
4
|
+
(what holds), the prescription in `AGENTS.md` (what to do and where), and the
|
|
5
|
+
proof in `test/` (pins that fail when the law is broken).
|
|
6
6
|
|
|
7
7
|
## Routing — what to read for each task
|
|
8
8
|
|
|
@@ -44,4 +44,5 @@ Two rules in `attention.ts` encode "exact decides" and must not be flattened:
|
|
|
44
44
|
|
|
45
45
|
Add a tier to the shared family in `mind/match.ts` with a derived gate
|
|
46
46
|
(`geometry.ts`), never a private `score >= k` check. A new mechanism is a
|
|
47
|
-
`(matcher, direction, gate)` configuration over that family
|
|
47
|
+
`(matcher, direction, gate)` configuration over that family
|
|
48
|
+
(`docs/architecture/match-project.md`).
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
# Tempting but Wrong —
|
|
1
|
+
# Tempting but Wrong — 13 Traps
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
Thirteen shortcuts that look plausible and break an invariant. Each states what
|
|
4
4
|
not to do, why it fails, and what to do instead.
|
|
5
5
|
|
|
6
6
|
### 1. `score >= threshold` decides identity
|
|
@@ -76,7 +76,7 @@ not to do, why it fails, and what to do instead.
|
|
|
76
76
|
- **WHY:** `frameSlots` reports (contracted gaps tagged
|
|
77
77
|
substitution/insertion/deletion); `carriesFillers` judges;
|
|
78
78
|
`Precomputed.frames` inventories — elects nothing (`AGENTS §2` Cross-cutting
|
|
79
|
-
contracts — `docs/architecture/
|
|
79
|
+
contracts — `docs/architecture/match-project.md` § Frame reading).
|
|
80
80
|
- **CORRECT:** Report everything in the shared layer; apply
|
|
81
81
|
`substituteAll(contA, fillersA→fillersB)==contB` and
|
|
82
82
|
frame-dominance/`W`-reach/distinctness in the consumer (reference). Pinned by
|
|
@@ -88,8 +88,7 @@ not to do, why it fails, and what to do instead.
|
|
|
88
88
|
`depth[i]`/`dominates(depth, aligned)` to decide climb/IDF.
|
|
89
89
|
- **WHY:** They measure different things: global reach (minority discriminates,
|
|
90
90
|
powers climb/pooling) vs weave-local depth with `MIN_WEAVE=2` (what the local
|
|
91
|
-
cohort shares, powers CAST) (`
|
|
92
|
-
measures of commonality).
|
|
91
|
+
cohort shares, powers CAST) (`docs/architecture/commonality.md`).
|
|
93
92
|
- **CORRECT:** Climb/attention uses corpus-global;
|
|
94
93
|
`frame(i) ⇔ depth[i]>MIN_WEAVE ∧ dominates(depth[i],aligned)` for CAST. Pinned
|
|
95
94
|
by `test/50-cast-analog-consensus-floor.test.mjs` and
|
|
@@ -142,3 +141,32 @@ not to do, why it fails, and what to do instead.
|
|
|
142
141
|
`twoEndedSeat`); turns are API state in `mind/mind.ts`, not segmentation.
|
|
143
142
|
Pinned by `test/59-fold-invariance.test.mjs` and
|
|
144
143
|
`test/63-fold-invariants.test.mjs`.
|
|
144
|
+
|
|
145
|
+
### 13. Capping a combinatorial explosion instead of budgeting it
|
|
146
|
+
|
|
147
|
+
- **WRONG:** Answer a combinatorial explosion with a geometry-derived limit — a
|
|
148
|
+
cap on the pairs a sweep enumerates, the continuations a hop may offer, the
|
|
149
|
+
candidates a scan probes. A derived limit is the right cutoff for a DECISION;
|
|
150
|
+
used as the answer to explosion it is a short-circuit.
|
|
151
|
+
- **WHY:** It stops the computation silently. Reach is lost, the capability that
|
|
152
|
+
depended on it goes with it, and no test fails, because the tests were written
|
|
153
|
+
against the capped behaviour. Capping and removing the cap are both wrong:
|
|
154
|
+
capping truncates, removing lets the cost run, and the two failure modes hide
|
|
155
|
+
each other.
|
|
156
|
+
- **CORRECT:** BUDGET it. The work is charged in the one currency
|
|
157
|
+
(`MICRO`/`STEP`/`CONCEPT`/`PASS`; `weight = moves + PASS·unaccounted`, see
|
|
158
|
+
`docs/architecture/cost-model.md`), the charge is visible in the meter and the
|
|
159
|
+
rationale, and the SEARCH decides whether the work is worth paying — so
|
|
160
|
+
inference is never locked by a limit and nothing is truncated in silence.
|
|
161
|
+
Where the work is mechanical rather than evidential — enumeration, scans,
|
|
162
|
+
sweeps — the answer is an algorithm whose cost is structural in the bytes it
|
|
163
|
+
is given, not a smaller cap.
|
|
164
|
+
- **THE IDEAL:** a universal **closure engine** — one law of closure, stated in
|
|
165
|
+
the quantities the machine already has (`leadsSomewhere`, the
|
|
166
|
+
exact-then-canonical identity, `accounted` bytes, the ladder, `hubBound`),
|
|
167
|
+
from which the reach of a gap, the offer of a hop, the depth of a join and the
|
|
168
|
+
scope of a substitution are CONSEQUENCES, not four separate decisions. Nothing
|
|
169
|
+
in this repository is that today.
|
|
170
|
+
- **THE STANDARD A CHANGE MUST MEET:** state which consequence it is, and show
|
|
171
|
+
it following from the law. A change that cannot be stated that way is not
|
|
172
|
+
ready.
|
package/docs/harness/gates.md
CHANGED
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
Four executable gates. Each: run the command, check what it guards, follow its
|
|
4
4
|
§.
|
|
5
5
|
|
|
6
|
-
## 1 — Correctness (all
|
|
6
|
+
## 1 — Correctness (all 90 suites)
|
|
7
7
|
|
|
8
8
|
```bash
|
|
9
9
|
npm test
|
|
@@ -13,8 +13,8 @@ Guards honest silence, determinism, and every pinned contract. Silence:
|
|
|
13
13
|
unrelated queries ground to nothing (`test/28`, `50`, `56`, `67`, `76`, `84`).
|
|
14
14
|
Determinism: same seed + deposit order + query gives byte-identical answer
|
|
15
15
|
(`test/20`). Every invariant is pinned — a simplification that fails a test is
|
|
16
|
-
wrong until the test is shown wrong. §14–25 (pipeline), §
|
|
17
|
-
AGENTS.md §2 invariants 1–5.
|
|
16
|
+
wrong until the test is shown wrong. §14–25 (pipeline), §64 (derived
|
|
17
|
+
thresholds), AGENTS.md §2 invariants 1–5.
|
|
18
18
|
|
|
19
19
|
## 2 — Work accounting (profiler)
|
|
20
20
|
|
|
@@ -27,8 +27,8 @@ Guards without trace: counters deterministic and diffable between runs; phases
|
|
|
27
27
|
nest (not disjoint — `think` contains every mechanism phase); shared analyses
|
|
28
28
|
charged to themselves, not to the first toucher; millisecond fields are
|
|
29
29
|
non-deterministic hints only. With `--trace`, recognition idempotence still
|
|
30
|
-
holds (`test/42`). `src/meter.ts`, `docs/architecture/meter.md`, §
|
|
31
|
-
§
|
|
30
|
+
holds (`test/42`). `src/meter.ts`, `docs/architecture/meter.md`, §55,
|
|
31
|
+
`AGENTS.md` §6.
|
|
32
32
|
|
|
33
33
|
## 3 — Dependency footprint
|
|
34
34
|
|
|
@@ -38,7 +38,7 @@ node --test test/88-dependency-footprint.test.mjs
|
|
|
38
38
|
|
|
39
39
|
Guards `dist/src` imports only `node:` + relative paths, and `package.json`
|
|
40
40
|
declares no `dependencies` (examples use `devDependencies` lazily). The
|
|
41
|
-
near-zero footprint is a product feature. AGENTS.md §
|
|
41
|
+
near-zero footprint is a product feature. AGENTS.md §7, §3 (store has one
|
|
42
42
|
runtime dep: `node:sqlite`).
|
|
43
43
|
|
|
44
44
|
## 4 — Fold invariance and sublinear scaling
|
|
@@ -53,4 +53,4 @@ positional; grid regression (14.3% survival) cannot pass. `14` — inference cos
|
|
|
53
53
|
is sublinear in corpus size (power-law exponent ≪ 1) and constant-rate in input
|
|
54
54
|
length; measured on independent disjoint corpora via log–log slope.
|
|
55
55
|
`src/geometry.ts` (`contentLevels`), `docs/architecture/fold-contract.md` +
|
|
56
|
-
`bounded-reads.md`, §
|
|
56
|
+
`bounded-reads.md`, §59, §63.
|
|
@@ -4,8 +4,8 @@
|
|
|
4
4
|
// the cache ceiling, the read budgets, the caps. A knob that describes ONE
|
|
5
5
|
// CORPUS (which pairs of SmolSent, how many SODA dialogues, how long an Aya
|
|
6
6
|
// field may be) belongs next to that corpus's adapter, together with the
|
|
7
|
-
// evidence that fixed its default —
|
|
8
|
-
// constraint
|
|
7
|
+
// evidence that fixed its default — a comment carries the constraint, and a
|
|
8
|
+
// constraint is only readable beside the code it constrains.
|
|
9
9
|
|
|
10
10
|
import { join } from "node:path";
|
|
11
11
|
|
|
@@ -37,7 +37,7 @@ import { convertedParquetUnits } from "./converted-parquet.js";
|
|
|
37
37
|
//
|
|
38
38
|
// So it displaces some wrong answers and manufactures others, INCLUDING turning
|
|
39
39
|
// a correct silence into a wrong answer — and honest silence is a stated
|
|
40
|
-
// property of this engine (
|
|
40
|
+
// property of this engine (INVARIANTS.md). On the mixed-curriculum store the
|
|
41
41
|
// same shape produced the fragment "nus" for "wake me up at nine am".
|
|
42
42
|
//
|
|
43
43
|
// That evidence is four probes on toy stores and is NOT conclusive; it is,
|
|
@@ -13,7 +13,7 @@
|
|
|
13
13
|
//
|
|
14
14
|
// THE ONLY THIRD-PARTY CODE IN THIS REPOSITORY IS BELOW, and it is LAZILY
|
|
15
15
|
// LOADED. Sema itself imports nothing outside `node:` — that is a product
|
|
16
|
-
// property, not an accident (AGENTS.md §
|
|
16
|
+
// property, not an accident (AGENTS.md §7) — and this trainer is an EXAMPLE,
|
|
17
17
|
// not part of the library. hyparquet (+ its Snappy codec) is therefore a dev
|
|
18
18
|
// dependency, and it is loaded by a dynamic import the first time a Parquet
|
|
19
19
|
// corpus is actually read: a curriculum with no Parquet stage (SmolSent,
|
package/jsr.json
CHANGED
package/package.json
CHANGED
package/src/config.ts
CHANGED
|
@@ -116,6 +116,23 @@ export interface MindConfig {
|
|
|
116
116
|
seed: number;
|
|
117
117
|
recallQueryK: number;
|
|
118
118
|
haloQueryK: number;
|
|
119
|
+
/** Corpus reading (see src/mind/corpus.ts): results per call, resolved
|
|
120
|
+
* nodes climbed from, contexts requested per climb, probes used to stride
|
|
121
|
+
* the id space when browsing, bytes of each side a preview keeps, and the
|
|
122
|
+
* smallest deposited note browsing will show. Capacities and budgets only —
|
|
123
|
+
* the one material floor (a resolved node must account for W bytes) is
|
|
124
|
+
* derived from the geometry, not declared here. */
|
|
125
|
+
corpusLimitMax: number;
|
|
126
|
+
corpusClimbs: number;
|
|
127
|
+
corpusContextsPerClimb: number;
|
|
128
|
+
corpusSampleProbes: number;
|
|
129
|
+
corpusPreviewBytes: number;
|
|
130
|
+
corpusSampleFloorBytes: number;
|
|
131
|
+
/** Items one rationale step may ITEMISE (the whole field is still counted in
|
|
132
|
+
* the step's note). A capacity of the rationale, not of recall: sharing
|
|
133
|
+
* `recallQueryK` meant `new Mind({recallQueryK: 100000})` un-bounded the very
|
|
134
|
+
* payload the bound exists for (found by an adversarial review). */
|
|
135
|
+
rationaleSampleK: number;
|
|
119
136
|
normalizeEpsilon: number;
|
|
120
137
|
cosineEpsilon: number;
|
|
121
138
|
|
|
@@ -131,6 +148,13 @@ export const DEFAULT_CONFIG: MindConfig = {
|
|
|
131
148
|
seed: 42,
|
|
132
149
|
recallQueryK: 12,
|
|
133
150
|
haloQueryK: 12,
|
|
151
|
+
rationaleSampleK: 12,
|
|
152
|
+
corpusLimitMax: 24,
|
|
153
|
+
corpusClimbs: 24,
|
|
154
|
+
corpusContextsPerClimb: 6,
|
|
155
|
+
corpusSampleProbes: 6000,
|
|
156
|
+
corpusPreviewBytes: 220,
|
|
157
|
+
corpusSampleFloorBytes: 12,
|
|
134
158
|
normalizeEpsilon: 1e-12,
|
|
135
159
|
cosineEpsilon: 1e-12,
|
|
136
160
|
alu: {
|
|
@@ -172,6 +196,17 @@ export function resolveConfig(opts: Partial<MindConfig> = {}): MindConfig {
|
|
|
172
196
|
seed: opts.seed ?? DEFAULT_CONFIG.seed,
|
|
173
197
|
recallQueryK: opts.recallQueryK ?? DEFAULT_CONFIG.recallQueryK,
|
|
174
198
|
haloQueryK: opts.haloQueryK ?? DEFAULT_CONFIG.haloQueryK,
|
|
199
|
+
rationaleSampleK: opts.rationaleSampleK ?? DEFAULT_CONFIG.rationaleSampleK,
|
|
200
|
+
corpusLimitMax: opts.corpusLimitMax ?? DEFAULT_CONFIG.corpusLimitMax,
|
|
201
|
+
corpusClimbs: opts.corpusClimbs ?? DEFAULT_CONFIG.corpusClimbs,
|
|
202
|
+
corpusContextsPerClimb: opts.corpusContextsPerClimb ??
|
|
203
|
+
DEFAULT_CONFIG.corpusContextsPerClimb,
|
|
204
|
+
corpusSampleProbes: opts.corpusSampleProbes ??
|
|
205
|
+
DEFAULT_CONFIG.corpusSampleProbes,
|
|
206
|
+
corpusPreviewBytes: opts.corpusPreviewBytes ??
|
|
207
|
+
DEFAULT_CONFIG.corpusPreviewBytes,
|
|
208
|
+
corpusSampleFloorBytes: opts.corpusSampleFloorBytes ??
|
|
209
|
+
DEFAULT_CONFIG.corpusSampleFloorBytes,
|
|
175
210
|
normalizeEpsilon: opts.normalizeEpsilon ?? DEFAULT_CONFIG.normalizeEpsilon,
|
|
176
211
|
cosineEpsilon: opts.cosineEpsilon ?? DEFAULT_CONFIG.cosineEpsilon,
|
|
177
212
|
alu: {
|