@hviana/sema 0.9.1 → 0.9.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -7,13 +7,15 @@
7
7
  // project) live in match.ts — the elementary match-and-project operation.
8
8
  import { cosine } from "../vec.js";
9
9
  import { gistOf, read } from "./primitives.js";
10
- import { canonicalWindows, leafIdPrefix, leafIdRun } from "./canonical.js";
10
+ import { canonicalWindows, chainReach, leafIdPrefix, leafIdRun, } from "./canonical.js";
11
11
  // Imported at the TOP, where every other import is. They used to sit 800 lines
12
12
  // down under a note claiming the position mattered ("before trace module is
13
13
  // loaded") — it does not: an ES module's static imports are HOISTED, so the
14
14
  // file's line order never decides load order. The note described an intention
15
15
  // the runtime does not honour; the imports move and the claim goes.
16
16
  import { decodeText } from "./rationale.js";
17
+ import { latin1 } from "../bytes.js";
18
+ import { windowIndex, witness } from "./evidence.js";
17
19
  //
18
20
  // Budgeted on the same terms as the reach memo below (caches.md): these three
19
21
  // maps are cleared on every write, but a long read-only session over a large
@@ -599,8 +601,10 @@ export function guidedNext(ctx, node) {
599
601
  ctx._edgeChoice.set(node, pick ?? -1);
600
602
  return pick;
601
603
  }
602
- /** Disambiguate among a node's learnt continuations by distributional
603
- * support. NOTE the `guide` contract: its VALUE is deliberately unused —
604
+ /** Disambiguate among a node's learnt continuations: first by the question's
605
+ * own witness of an establishing context (the exact tier, see
606
+ * {@link askedContinuations}), then by distributional support. NOTE the
607
+ * `guide` contract: its VALUE is deliberately unused —
604
608
  * only its PRESENCE gates disambiguation (a null guide means no query is in
605
609
  * flight, so structural walkers keep plain first-edge behaviour). The
606
610
  * gist-cosine of short answer candidates against a query guide is dominated
@@ -621,13 +625,30 @@ export function chooseNext(ctx, id, guide) {
621
625
  return undefined;
622
626
  if (nx.length === 1 || !guide)
623
627
  return nx[0];
628
+ // THE EXACT TIER — the continuation the QUESTION names. Every other
629
+ // disambiguation below reads popularity, and is right to refuse the gist (see
630
+ // the doc above); but the corpus also wrote down, for each continuation,
631
+ // WHICH QUESTIONS IT ANSWERS — its establishing contexts — and a question is
632
+ // not a gist. When one of them is witnessed by the asker's bytes plus the
633
+ // node's own, that continuation is the one being asked for. Measured on the
634
+ // 31.7M-node store: `Who is the father of Frederick II?` answered the
635
+ // citizenship fact (the most-poured of eight) while `Frederick II father` —
636
+ // one of the father fact's own establishing contexts — lay wholly inside the
637
+ // question. Exact, so it ranks first (exact-vs-approximate.md); when nothing
638
+ // is witnessed the ladder below decides exactly as before.
639
+ const asked = ctx._edgeAsked;
640
+ const named = asked === null ? null : askedContinuations(ctx, id, nx, asked);
641
+ if (named !== null && named.length === 1)
642
+ return named[0];
624
643
  // Cap candidates at √N — the same bound the original chooseAmong used.
625
644
  // A hub context can accumulate thousands of continuations; the best-fit
626
645
  // one is among the first √N by insertion order (edges are never deleted,
627
646
  // so the oldest are the most established). A strongly-supported edge
628
647
  // inserted beyond the cap is invisible here — the deliberate trade
629
- // against paying O(fan-out) count reads on every disambiguation.
630
- const capped = nx; // already the hub-capped prefix, by the read above
648
+ // against paying O(fan-out) count reads on every disambiguation. Several
649
+ // continuations named EQUALLY by the question are told apart by the same
650
+ // ladder, over them alone.
651
+ const capped = named ?? nx; // already the hub-capped prefix, by the read above
631
652
  // Distributional-evidence disambiguation, consulting BOTH read-outs of the
632
653
  // evidence the training poured:
633
654
  // 1. prevCount — how many DISTINCT contexts predict this candidate (one
@@ -691,6 +712,184 @@ export function chooseNext(ctx, id, guide) {
691
712
  }
692
713
  return best;
693
714
  }
715
+ /** The response canonicalizer's reading of `bytes` when it keeps every offset
716
+ * — so a window found in the canonical bytes sits at the same place in the
717
+ * asker's — else the bytes themselves. Text canon is offset-preserving on
718
+ * ASCII without interior whitespace runs; where it is not, the raw bytes are
719
+ * read and a case-variant window simply does not match. */
720
+ export function offsetCanon(ctx, bytes) {
721
+ if (ctx.canon === null)
722
+ return bytes;
723
+ const c = ctx.canon(bytes);
724
+ return c.length === bytes.length ? c : bytes;
725
+ }
726
+ /** The continuations of `id` that `asked` NAMES (see {@link
727
+ * askedContinuations}) — for a caller holding material other than the whole
728
+ * question: the multi-hop walk asks with what of the question no product has
729
+ * restated yet. null when none is named. */
730
+ export function namedContinuations(ctx, id, asked) {
731
+ const nx = ctx.store.nextFirst(id, hubBound(ctx));
732
+ return nx.length === 0 ? null : askedContinuations(ctx, id, nx, asked);
733
+ }
734
+ /** The continuations of `id` (among `nx`) that the question NAMES: one of
735
+ * their establishing contexts — a predecessor other than `id` itself — is
736
+ * wholly witnessed by the question plus `id`'s own bytes (evidence.ts), with
737
+ * the question supplying at least one window the node does not. Ranked by
738
+ * how much of the question witnesses it; null when none is named.
739
+ *
740
+ * THE NODE'S OWN BYTES ARE MATERIAL because a derivation stands on them. On
741
+ * the second hop of `Where was the place of death of the director of film
742
+ * Beat Girl?` the node is `Edmond T. Gréville` — reached, never written — and
743
+ * its fact's establishing context `Edmond T. Gréville place of death` is held
744
+ * by neither the question nor the node, only by both. Measured over 5,236
745
+ * held-out 2Wiki questions: such a context is wholly witnessed by the
746
+ * question alone 69 times, by the question and the first hop 2,153 times.
747
+ *
748
+ * BOUNDED: predecessor reads share one √N budget across the candidates (see
749
+ * the loop below), asked cheapest first; past it the tier abstains for the
750
+ * rest, metered, and the ladder decides as before.
751
+ * Each form is read at most to the material's length — a form longer than
752
+ * everything at hand cannot be wholly witnessed without repeating it. */
753
+ function askedContinuations(ctx, id, nx, asked) {
754
+ return askedEntry(ctx, id, nx, asked).named;
755
+ }
756
+ /** The question spans that NAMED `pick` among `id`'s continuations — the
757
+ * evidence a projection through that pick stands on, so a mechanism can
758
+ * account for what the question said about it (mechanism-market.md:
759
+ * evidence travels). Empty when the question names no continuation of `id`
760
+ * or names others. */
761
+ export function askedEvidence(ctx, id, pick) {
762
+ const asked = ctx._edgeAsked;
763
+ if (asked === null)
764
+ return [];
765
+ const nx = ctx.store.nextFirst(id, hubBound(ctx));
766
+ const entry = askedEntry(ctx, id, nx, asked);
767
+ return entry.named?.includes(pick) ? entry.spans.get(pick) ?? [] : [];
768
+ }
769
+ function askedEntry(ctx, id, nx, asked) {
770
+ let memo = askedMemo.get(asked);
771
+ if (memo === undefined)
772
+ askedMemo.set(asked, memo = new Map());
773
+ const hit = memo.get(id);
774
+ if (hit !== undefined && !ctx.trace)
775
+ return hit;
776
+ const entry = askedContinuationsImpl(ctx, id, nx, asked);
777
+ memo.set(id, entry);
778
+ return entry;
779
+ }
780
+ /** One pick per node per question — every mechanism of a response asks the
781
+ * same node about the same question (the guided-pick memo's own reason). */
782
+ const askedMemo = new WeakMap();
783
+ function askedContinuationsImpl(ctx, id, nx, asked) {
784
+ const none = { named: null, spans: new Map() };
785
+ const W = ctx.space.maxGroup;
786
+ // A SATURATED READ IS NOT A CANDIDATE SET. When the continuations came back
787
+ // at the √N cap the read may have cut the named one off, so "none of these is
788
+ // named" and "this is the named one" are both unfounded — and this is exactly
789
+ // where witnessing would read most. The tier abstains, metered, and the
790
+ // distributional ladder decides as it always has.
791
+ if (nx.length >= hubBound(ctx)) {
792
+ if (ctx.meter)
793
+ ctx.meter.askedReadsSaturated++;
794
+ return none;
795
+ }
796
+ const cache = getStructCache(ctx);
797
+ const ownCap = asked.bytes.length * W;
798
+ const own = offsetCanon(ctx, read(ctx, id, ownCap));
799
+ const ownIndex = windowIndex(own, W);
800
+ // Naming needs the question to say at least one window the node does not:
801
+ // when the node already holds every window of the question (the question IS
802
+ // this context, or a piece of it), nothing can be named, and nothing is read.
803
+ let beyond = false;
804
+ for (const key of asked.index.keys()) {
805
+ if (!ownIndex.has(key)) {
806
+ beyond = true;
807
+ break;
808
+ }
809
+ }
810
+ if (!beyond)
811
+ return none;
812
+ const indexes = [asked.index, ownIndex];
813
+ const formCap = asked.bytes.length + own.length;
814
+ // BOUNDED READS (bounded-reads.md): the decision reads at most √N
815
+ // establishing contexts — floored at the write side's own arity `chainReach(W)`
816
+ // so a store too small for √N to cover one fact's questions still decides.
817
+ // Candidates are asked CHEAPEST FIRST (fewest establishing contexts): a common
818
+ // reply established by hundreds of contexts would otherwise spend the whole
819
+ // allowance alone. The order changes what is READ, never what wins: scores
820
+ // are compared afterwards in the continuations' own order.
821
+ let budget = Math.max(hubBound(ctx), chainReach(W));
822
+ const order = nx
823
+ .map((n, at) => ({ n, at, support: cachedPrevCount(ctx, n, cache) }))
824
+ .filter((c) => c.support >= 2) // only `id` establishes the rest
825
+ .sort((a, b) => a.support - b.support || a.at - b.at);
826
+ const scored = [];
827
+ for (const { n, at, support } of order) {
828
+ if (support > budget) {
829
+ if (ctx.meter)
830
+ ctx.meter.askedReadsSaturated++;
831
+ break;
832
+ }
833
+ budget -= support;
834
+ if (ctx.meter)
835
+ ctx.meter.askedPredecessorReads += support;
836
+ let score = 0;
837
+ let by = -1;
838
+ let spans = [];
839
+ for (const c of ctx.store.prevFirst(n, support)) {
840
+ if (c === id)
841
+ continue;
842
+ // The form's FIRST window decides most refusals: one short prefix read
843
+ // before the whole form is reconstructed (a conversation-length
844
+ // predecessor would otherwise be read in full to fail on its opening).
845
+ const head = offsetCanon(ctx, read(ctx, c, W));
846
+ if (head.length < W)
847
+ continue;
848
+ if (!indexes.some((ix) => ix.has(latin1(head))))
849
+ continue;
850
+ const form = read(ctx, c, formCap + 1);
851
+ if (form.length < W || form.length > formCap)
852
+ continue;
853
+ const w = witness(offsetCanon(ctx, form), indexes, W);
854
+ if (!w.complete || w.bytes < W)
855
+ continue;
856
+ if (w.bytes > score) {
857
+ score = w.bytes;
858
+ by = c;
859
+ spans = w.spans;
860
+ }
861
+ }
862
+ if (score > 0)
863
+ scored.push({ n, at, score, by, spans });
864
+ }
865
+ scored.sort((a, b) => a.at - b.at);
866
+ let best = [];
867
+ let bestBytes = 0;
868
+ let witnessed = null;
869
+ for (const { n, score, by } of scored) {
870
+ if (score > bestBytes) {
871
+ best = [n];
872
+ bestBytes = score;
873
+ witnessed = by;
874
+ }
875
+ else if (score === bestBytes)
876
+ best.push(n);
877
+ }
878
+ if (best.length === 0)
879
+ return none;
880
+ if (ctx.meter)
881
+ ctx.meter.askedContinuations++;
882
+ if (ctx.trace && witnessed !== null) {
883
+ ctx.trace.step("askedContinuation", [rItemShort(ctx, id, "node"), rItemShort(ctx, witnessed, "asked")], best.map((n) => rItemShort(ctx, n, "named")), `${nx.length} continuations — the question witnesses ` +
884
+ `${best.length === 1 ? "one's" : `${best.length}'`} own establishing ` +
885
+ `context (${bestBytes} question byte(s) beyond the node)`);
886
+ }
887
+ const evidence = new Map();
888
+ for (const c of scored)
889
+ if (best.includes(c.n))
890
+ evidence.set(c.n, c.spans);
891
+ return { named: best, spans: evidence };
892
+ }
694
893
  /** The perceived gist of a candidate node, through the session gist cache.
695
894
  * Re-gisting a candidate is a full river fold of its bytes — the measured
696
895
  * recall bottleneck (a hub context offers up to √N continuations, EACH
@@ -776,6 +975,84 @@ export function allWindowsAreScaffolding(ctx, query) {
776
975
  }
777
976
  return sawOne;
778
977
  }
978
+ /** Per offset of `bytes`: 1 when the W-window there is a stored form contained
979
+ * in more than √N places (corpus-global scaffolding; see the floor below),
980
+ * else 0. Memoised per
981
+ * byte array for the life of the store's read-only response. */
982
+ export function hubWindows(ctx, bytes) {
983
+ const hit = hubWindowMemo.get(bytes);
984
+ if (hit !== undefined)
985
+ return hit;
986
+ const W = ctx.space.maxGroup;
987
+ // Floored at the write side's arity: inside ONE deposit's fold a window is
988
+ // already contained by up to `chainReach(W)` chunks and branches, so on a
989
+ // store of a few facts the √N reading would call every window frame — that
990
+ // is fold structure, not corpus commonality.
991
+ const bound = Math.max(hubBound(ctx), chainReach(W));
992
+ const hub = new Uint8Array(Math.max(0, bytes.length - W + 1));
993
+ for (let o = 0; o < hub.length; o++) {
994
+ const ids = leafIdRun(ctx, bytes, o, o + W);
995
+ const id = ids === null ? null : ctx.store.findBranch(ids);
996
+ if (id !== null && ctx.store.containersSlice(id, bound, 1).length > 0) {
997
+ hub[o] = 1;
998
+ }
999
+ }
1000
+ hubWindowMemo.set(bytes, hub);
1001
+ return hub;
1002
+ }
1003
+ const hubWindowMemo = new WeakMap();
1004
+ /** The query's SCAFFOLDING CORE, as spans: the bytes every W-window over which
1005
+ * is a hub (see {@link hubWindows}) — what is nothing but frame, where
1006
+ * {@link scaffoldExtents} is what a frame window reaches. */
1007
+ export function scaffoldSpans(ctx, query) {
1008
+ const W = ctx.space.maxGroup;
1009
+ const hub = hubWindows(ctx, query);
1010
+ const n = hub.length;
1011
+ if (n <= 0)
1012
+ return [];
1013
+ const spans = [];
1014
+ let start = -1;
1015
+ for (let i = 0; i < query.length; i++) {
1016
+ let all = true;
1017
+ for (let o = Math.max(0, i - W + 1); o <= Math.min(i, n - 1); o++) {
1018
+ if (!hub[o]) {
1019
+ all = false;
1020
+ break;
1021
+ }
1022
+ }
1023
+ if (all && start < 0)
1024
+ start = i;
1025
+ if (!all && start >= 0) {
1026
+ spans.push([start, i]);
1027
+ start = -1;
1028
+ }
1029
+ }
1030
+ if (start >= 0)
1031
+ spans.push([start, query.length]);
1032
+ return spans;
1033
+ }
1034
+ /** The EXTENTS of the query's SCAFFOLDING windows, merged: every byte some
1035
+ * W-window reaches that is a stored form contained in more than √N places —
1036
+ * corpus-global commonality (commonality.md), the same "hub" reading as
1037
+ * {@link allWindowsAreScaffolding} and the bridge's `explainedSpan`. A window
1038
+ * the store never saw is NOT scaffolding. The extent, not the core, is what
1039
+ * a coverage test needs: a span that still holds one hub window can be
1040
+ * "carried" by any fact that holds that window. */
1041
+ export function scaffoldExtents(ctx, query) {
1042
+ const W = ctx.space.maxGroup;
1043
+ const hub = hubWindows(ctx, query);
1044
+ const spans = [];
1045
+ for (let o = 0; o < hub.length; o++) {
1046
+ if (!hub[o])
1047
+ continue;
1048
+ const last = spans[spans.length - 1];
1049
+ if (last !== undefined && o <= last[1])
1050
+ last[1] = o + W;
1051
+ else
1052
+ spans.push([o, o + W]);
1053
+ }
1054
+ return spans;
1055
+ }
779
1056
  // ── THE PREFIX SUPPLY ───────────────────────────────────────────────────────
780
1057
  //
781
1058
  // A RETRIEVAL capability, not a grounding one: "which trained forms does this
@@ -7,6 +7,7 @@ import type { MindConfig } from "../config.js";
7
7
  import type { Meter } from "../meter.js";
8
8
  import type { GraphSearch, Leaf, Seg, Site } from "./graph-search.js";
9
9
  import type { Rationale } from "./rationale.js";
10
+ import type { WindowIndex } from "./evidence.js";
10
11
  import type { ContentFold, Grid } from "../geometry.js";
11
12
  /** One {@link MindContext._depositTrees} entry — see that field's doc.
12
13
  *
@@ -370,6 +371,14 @@ export interface MindContext extends GraphSearchHost {
370
371
  * ordinary respond() and for the first turn of a conversation. */
371
372
  currentTurnStart: number;
372
373
  _edgeGuide: Vec | null;
374
+ /** The question currently being answered, as the material `chooseNext`'s
375
+ * exact tier witnesses a continuation's establishing contexts against —
376
+ * its bytes (canonical when the response's canon preserves offsets) and
377
+ * their window index. Set and cleared with `_edgeGuide`. */
378
+ _edgeAsked: {
379
+ bytes: Uint8Array;
380
+ index: WindowIndex;
381
+ } | null;
373
382
  _edgeChoice: Map<number, number>;
374
383
  _prevSeen: Set<number> | null;
375
384
  /** Session cache of node-id → perceived gist, for candidate scoring
package/docs/INDEX.md CHANGED
@@ -19,24 +19,25 @@ proof in `test/` (pins that fail when the law is broken).
19
19
  | Change vector search | `docs/architecture/exact-vs-approximate.md` + `docs/architecture/bounded-reads.md` | Scores propose, bytes dispose; ANN is bounded by `hubBound` |
20
20
  | Profile or bound work | `docs/architecture/meter.md` + `docs/architecture/bounded-reads.md` | `meter.ts` is write-only; counters are product, phases are hints |
21
21
 
22
- ## Architecture laws (14)
22
+ ## Architecture laws (15)
23
23
 
24
- | Law | File | Summary | Pins |
25
- | --- | ------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------- |
26
- | 1 | `docs/architecture/determinism.md` | No `Math.random`/`Date.now` in behaviour; seed-derived randomness; corpus-determined tie-breaks | `test/20` |
27
- | 2 | `docs/architecture/thresholds.md` | Every decision cutoff derived in `geometry.ts` over D/W/N; no tunable knobs | `test/40`, `test/64` |
28
- | 3 | `docs/architecture/exact-vs-approximate.md` | Vector scores rank only; identity via content-addressed lookup; five graded ladders | `test/51` |
29
- | 4 | `docs/architecture/cost-model.md` | Single ladder `MICRO`/`STEP`/`CONCEPT`/`PASS`; weight `moves + PASS·unaccounted`; `STEP`-grade compare | `test/04`, `test/55` |
30
- | 5 | `docs/architecture/match-project.md` | Shared `match.ts` family (`locate`/`alignGraded`/`frameSlots`/`project`); voicing gates belong to consumers | `test/24`, `test/76` |
31
- | 6 | `docs/architecture/mechanism-market.md` | `PipelineMechanism` (`floor`/`run`/`parse`); admissible-floor pruning, investment discipline, run-ahead bounds | `test/01`, `test/04`, `test/153` |
32
- | 7 | `docs/architecture/commonality.md` | Three: global (`reachOf`+`dominates`), weave-local (`depth[]`), window rarity | `test/17`, `test/34` |
33
- | 8 | `docs/architecture/bounded-reads.md` | No per-query read grows with N; `hubBound=√N` enforced at store via LIMIT/probe/prefix caps | `test/77`, `test/90` |
34
- | 9 | `docs/architecture/store.md` | `AbstractStore` owns dedup/indexing/batch; `store-sqlite.ts` is thin wrappers; canon index optional | `test/08` |
35
- | 10 | `docs/architecture/fold-contract.md` | `perceiveDeposit` and `perceive` agree; the read side names a branch as `intern` does; `contentLevels` is single boundary rule; no W/offset dependence | `test/59`, `test/63`, `test/148`, `test/152` |
36
- | 11 | `docs/architecture/memoization.md` | `Precomputed` is per-response lazy cache (promise-cached async); `beginResponse`/`endResponse` lifecycle | `test/42` |
37
- | 12 | `docs/architecture/saturation.md` | Every walk names a deciding saturation beside its cap; cap is safety net, not decision | `test/27`, `test/16` |
38
- | 13 | `docs/architecture/meter.md` | `meter.ts` is write-only work accounting; counts are exact, phases nest | `test/55` |
39
- | 14 | `docs/architecture/closure.md` | A derivation is closed when its structure accounts for the question's remainder; every transition asks that law, one engine walks the layers | `test/133`–`151` |
24
+ | Law | File | Summary | Pins |
25
+ | --- | ------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------- |
26
+ | 1 | `docs/architecture/determinism.md` | No `Math.random`/`Date.now` in behaviour; seed-derived randomness; corpus-determined tie-breaks | `test/20` |
27
+ | 2 | `docs/architecture/thresholds.md` | Every decision cutoff derived in `geometry.ts` over D/W/N; no tunable knobs | `test/40`, `test/64` |
28
+ | 3 | `docs/architecture/exact-vs-approximate.md` | Vector scores rank only; identity via content-addressed lookup; six graded ladders | `test/51` |
29
+ | 4 | `docs/architecture/cost-model.md` | Single ladder `MICRO`/`STEP`/`CONCEPT`/`PASS`; weight `moves + PASS·unaccounted`; `STEP`-grade compare | `test/04`, `test/55` |
30
+ | 5 | `docs/architecture/match-project.md` | Shared `match.ts` family (`locate`/`alignGraded`/`frameSlots`/`project`); voicing gates belong to consumers | `test/24`, `test/76` |
31
+ | 6 | `docs/architecture/mechanism-market.md` | `PipelineMechanism` (`floor`/`run`/`parse`); admissible-floor pruning, investment discipline, run-ahead bounds | `test/01`, `test/04`, `test/153` |
32
+ | 7 | `docs/architecture/commonality.md` | Three: global (`reachOf`+`dominates`), weave-local (`depth[]`), window rarity | `test/17`, `test/34` |
33
+ | 8 | `docs/architecture/bounded-reads.md` | No per-query read grows with N; `hubBound=√N` enforced at store via LIMIT/probe/prefix caps | `test/77`, `test/90` |
34
+ | 9 | `docs/architecture/store.md` | `AbstractStore` owns dedup/indexing/batch; `store-sqlite.ts` is thin wrappers; canon index optional | `test/08` |
35
+ | 10 | `docs/architecture/fold-contract.md` | `perceiveDeposit` and `perceive` agree; the read side names a branch as `intern` does; `contentLevels` is single boundary rule; no W/offset dependence | `test/59`, `test/63`, `test/148`, `test/152` |
36
+ | 11 | `docs/architecture/memoization.md` | `Precomputed` is per-response lazy cache (promise-cached async); `beginResponse`/`endResponse` lifecycle | `test/42` |
37
+ | 12 | `docs/architecture/saturation.md` | Every walk names a deciding saturation beside its cap; cap is safety net, not decision | `test/27`, `test/16` |
38
+ | 13 | `docs/architecture/meter.md` | `meter.ts` is write-only work accounting; counts are exact, phases nest | `test/55` |
39
+ | 14 | `docs/architecture/closure.md` | A derivation is closed when its structure accounts for the question's remainder; every transition asks that law, one engine walks the layers | `test/133`–`151` |
40
+ | 15 | `docs/architecture/evidence.md` | A stored form is identified when the material at hand (question ∪ the node a derivation stands on) witnesses every byte of it, order-free; the question NAMES a continuation through its establishing context; a step it did not name pays from what is still owed | `test/154` |
40
41
 
41
42
  ## Mechanisms (8)
42
43
 
@@ -16,3 +16,4 @@
16
16
  | 12 | Meter contracts | `src/meter.ts:Meter,PhaseCost,time` `src/mind/pipeline-mechanism.ts:Precomputed.shared` | `test/55` | `meter.md` |
17
17
  | 13 | Saturation | `traverse.ts:edgeAncestors,types.ts:SaturationReason` `src/mind/junction.ts:junctionContainersFrom` `src/mind/resonance.ts:pivotInto` | `test/27` `test/16` | `saturation.md` |
18
18
  | 14 | Closure | `src/mind/derivation.ts:admissible,advance,closeOver` | `test/133`–`151` | `closure.md` |
19
+ | 15 | Witnessed evidence | `src/mind/evidence.ts:witness,windowIndex` `src/mind/traverse.ts:chooseNext,askedEvidence,namedContinuations,scaffoldExtents` | `test/154` | `evidence.md` |
@@ -0,0 +1,113 @@
1
+ # Witnessed Evidence — The Question Names the Step
2
+
3
+ > **Law:** a stored form is identified by the material at hand when every one of
4
+ > its bytes lies in a W-window that material holds — in any order, at any place,
5
+ > and wherever each piece of the material came from. A step the question did not
6
+ > name has to be paid for by material the question still owes.
7
+
8
+ ## The operation — `src/mind/evidence.ts`
9
+
10
+ `witness(form, indexes, W)` reads one form against a list of window indexes
11
+ (`windowIndex`). It is the order-free reading of correspondence, beside
12
+ `alignRuns` (which produces runs) and `junctionContainersFrom(…, unordered)`
13
+ (which finds containers). It is exact, deterministic and linear: one index per
14
+ source and one probe per window of the form. A window is credited to the LAST
15
+ source that holds it, so source 0 (the question) is credited only with what
16
+ nothing else at hand supplies. A form shorter than W is never witnessed.
17
+
18
+ ## Where the material comes from
19
+
20
+ The corpus records, for every continuation, the questions that establish it: its
21
+ predecessors. A 2Wiki fact `The father of Frederick II is Peter III of
22
+ Aragon.`
23
+ is established by `Frederick II` and by `Frederick II father`. The asker's
24
+ question rarely repeats either one byte for byte, but it often holds every byte
25
+ of one of them.
26
+
27
+ The derivation stands on more than the question. The node it is following is
28
+ material too. On the second hop of
29
+ `Where was the place of death of the
30
+ director of film Beat Girl?` the node is
31
+ `Edmond T. Gréville`, which the first hop reached and the asker never wrote. The
32
+ establishing question `Edmond T.
33
+ Gréville place of death` is held by neither the
34
+ question nor the first hop's fact, only by both. Measured over 5,236 held-out
35
+ 2WikiMultihopQA compositional questions, such a context is wholly witnessed by
36
+ the question alone 69 times, and by the question plus the first hop 2,153 times.
37
+
38
+ ## Its consumers
39
+
40
+ | Where | What it decides |
41
+ | --------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
42
+ | `chooseNext` (traverse.ts) | **The exact tier.** It names the continuation one of whose establishing contexts (other than the node) is witnessed by the question plus the node, with the question supplying at least one window the node does not. It ranks first among the readings; the distributional ladder decides only when nothing is named, or among continuations named equally. |
43
+ | `askedEvidence` (traverse.ts) | The question spans that named a pick. A mechanism projecting through the pick accounts for them, because evidence travels (mechanism-market.md). Recall's argument binding uses this. |
44
+ | `preConsumed` (pipeline.ts) | When a grounding does not declare `used`, what it spoke for is the forms inside its answer that the question already holds. The entity the answer added stays pivotable. |
45
+ | The walk (reasoning.ts) | A pivot is NAMED when one of its continuation's establishing contexts is witnessed by what of the question no product has said yet, plus the pivot. A named pivot MOVES. An unnamed one is offered only while the derivation still owes something, and the law then decides by carrying. |
46
+ | Cover sites (mechanisms/cover.ts) | A FRAGMENT is a form that sits inside other forms, has several continuations, and leaves at least one window of the question beyond it. It answers other questions, so it leads somewhere for this question only when the question names one of its continuations. |
47
+
48
+ A cover span made of nothing but scaffolding (every window a hub,
49
+ `scaffoldSpans`) is not accounted. The form recognised there, such as the song
50
+ `What` in the trained store, is one of thousands the bytes could name. This is
51
+ measured only on the trained store (`What country is Jerry Bock a citizen of?`
52
+ was won by `The performer of What is Melinda Marx.` glued onto the right fact).
53
+ A synthetic fixture could not reproduce that regime, because content addressing
54
+ folds every filler's `What` into a handful of shared nodes, so no suite test
55
+ pins this rule.
56
+
57
+ ## What the question owes
58
+
59
+ The derivation is born owing only its discriminative material. The bytes a
60
+ corpus-global scaffolding window reaches (`scaffoldExtents`, the same "hub"
61
+ reading as `allWindowsAreScaffolding` and the bridge's `explainedSpan`) are
62
+ nobody's debt. Otherwise a step could claim to pay `Who is the` by restating
63
+ `is`, which every fact holds. Pricing is untouched: the ladder still charges
64
+ every unexplained byte, because for a question made only of scaffolding,
65
+ covering those bytes is the evidence. Making scaffolding free in the market was
66
+ measured and refused, because it changed dialogue answers
67
+ (`How are you
68
+ today?`).
69
+
70
+ The hub reading behind `scaffoldExtents` is floored at `chainReach(W)`
71
+ containers. Inside one deposit's fold, a window is already contained by up to
72
+ that many chunks and branches, so on a store of a few facts the √N reading would
73
+ call every window frame. That count measures fold structure, not corpus
74
+ commonality (`test/22`'s two-fact chains are exactly that regime).
75
+
76
+ ## Bounds
77
+
78
+ The exact tier reads at most √N establishing contexts per decision, floored at
79
+ `chainReach(W)`. It asks the cheapest candidates first and compares scores
80
+ afterwards in the continuations' own order. It abstains, metered as
81
+ `askedReadsSaturated`, in two cases: the continuation read came back at the √N
82
+ cap, or the question holds no window the node lacks. One short prefix read
83
+ refuses most predecessors before a whole form is reconstructed. Picks are
84
+ memoized per node per question.
85
+
86
+ ## Measured
87
+
88
+ | Measure | Before | After |
89
+ | -------------------------------------------------------------------------------------- | ------ | ------ |
90
+ | 2Wiki held-out fixture (300 rows, deposited as `wiki2.ts` does), compositional correct | 13/133 | 47/133 |
91
+ | Same fixture, pivot steps | 3 | 90+ |
92
+ | Same fixture, inference correct | 8/37 | 6/37 |
93
+
94
+ The inference drop is real. Two hops of `father` from a question that says
95
+ `father` once, as in `paternal grandfather`, used to be reached by
96
+ over-extension and are no longer. The store holds no evidence that `grandfather`
97
+ composes `father` twice.
98
+
99
+ On the 31.7M-node store's 116-query battery, two answers became correct
100
+ (`What country is Jerry Bock a citizen of?` and
101
+ `What is the country of citizenship of Frederick II?`), two lost a junk
102
+ composition, three changed between wrong answers, and no correct answer was
103
+ lost. CPU was 150 s against two baseline runs of 139 s and 166 s, inside the
104
+ noise. The deterministic read counters rose about 20% (`bytesRead`,
105
+ `nodeRecords`), with ANN queries unchanged.
106
+
107
+ ## Pins
108
+
109
+ - `test/154` — witnessing semantics. The named continuation beats the
110
+ most-poured one. The second hop is named by the question plus the introduced
111
+ entity, whatever grounded the first hop. A question that names no further step
112
+ is not extended. A fragment voices none of its continuations unless the
113
+ question names one. Each assertion was verified by mutation.
@@ -1,4 +1,4 @@
1
- # Exact vs Approximate — The Law and Its Five Ladders
1
+ # Exact vs Approximate — The Law and Its Six Ladders
2
2
 
3
3
  Vector scores (`resonate` / `resonateHalo`) are RaBitQ **estimates**. They rank
4
4
  candidates and gate broad regions; they never decide identity. Identity is
@@ -15,17 +15,18 @@ thresholds gate breadth, not truth.
15
15
 
16
16
  ## Graded evidence ladders
17
17
 
18
- Five subsystems share one shape — **exact → distributional → geometric** — with
18
+ Six subsystems share one shape — **exact → distributional → geometric** — with
19
19
  earlier tiers strictly preferred. Never reorder tiers; never let an approximate
20
20
  tier override an exact one.
21
21
 
22
- | # | Site | Ladder (strong → weak) | File |
23
- | - | ------------------ | ------------------------------------------------------------------------------------------------------------------------ | ----------------------------------------- |
24
- | 1 | `resolve` | exact content-addressed fold → `canonResolve` (equivalence class, hash-then-verify) | `mind/primitives.ts` |
25
- | 2 | `locate` | exact bytes → halo role → gist | `mind/match.ts` |
26
- | 3 | `alignGraded` | literal W-gram runs → halo-matched sites + climb proposals (weave) | `mind/match.ts` / `pipeline-mechanism.ts` |
27
- | 4 | `bridge` | junction containers → edge → synonym → whole-gist | `mind/resonance.ts` |
28
- | 5 | `crossRegionVotes` | exact containers → single synonym → double → `structuralResonance` (synthetic gist, gated hardest — no byte containment) | `mind/attention.ts` |
22
+ | # | Site | Ladder (strong → weak) | File |
23
+ | - | ------------------ | --------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------- |
24
+ | 1 | `resolve` | exact content-addressed fold → `canonResolve` (equivalence class, hash-then-verify) | `mind/primitives.ts` |
25
+ | 2 | `locate` | exact bytes → halo role → gist | `mind/match.ts` |
26
+ | 3 | `alignGraded` | literal W-gram runs → halo-matched sites + climb proposals (weave) | `mind/match.ts` / `pipeline-mechanism.ts` |
27
+ | 4 | `bridge` | junction containers → edge → synonym → whole-gist | `mind/resonance.ts` |
28
+ | 5 | `crossRegionVotes` | exact containers → single synonym → double → `structuralResonance` (synthetic gist, gated hardest — no byte containment) | `mind/attention.ts` |
29
+ | 6 | `chooseNext` | an establishing context witnessed by the question ∪ the node (evidence.md) → distributional support (prevCount, poured mass) → first-inserted | `mind/traverse.ts` |
29
30
 
30
31
  ## Asymmetries (attention)
31
32
 
package/jsr.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "$schema": "https://jsr.io/schema/config-file.v1.json",
3
3
  "name": "@hviana/sema",
4
- "version": "0.9.1",
4
+ "version": "0.9.2",
5
5
  "exports": "./src/index.ts"
6
6
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@hviana/sema",
3
- "version": "0.9.1",
3
+ "version": "0.9.2",
4
4
  "description": "Sema: a non-parametric, instance-based reasoning system.",
5
5
  "repository": {
6
6
  "type": "git",
package/src/meter.ts CHANGED
@@ -265,6 +265,19 @@ export class Meter {
265
265
  /** Times the reasoner pivoted on a span its answer contains and stepped
266
266
  * across that fact. */
267
267
  pivotSteps = 0;
268
+ /** `chooseNext` picks the question NAMED — a continuation one of whose own
269
+ * establishing contexts the question (plus the node) wholly witnesses
270
+ * (traverse.ts, the exact tier). */
271
+ askedContinuations = 0;
272
+ /** Predecessor rows that exact tier read, against its shared √N budget. */
273
+ askedPredecessorReads = 0;
274
+ /** Times the tier abstained on a read it could not trust: the continuations
275
+ * came back at the √N cap, or its predecessor budget ran out before every
276
+ * continuation was asked about — the distributional ladder decided. */
277
+ askedReadsSaturated = 0;
278
+ /** Cover sites dropped as FRAGMENTS whose several continuations the question
279
+ * names none of (mechanisms/cover.ts). */
280
+ unaskedFragments = 0;
268
281
  /** Canon probes REFUSED because the canon budget ran out — the one thing the
269
282
  * budget does that nothing could see. The budget itself is derived
270
283
  * (`bytes.length · chainReach(W)²`, recognition.ts), and the cheap exact route