@hviana/sema 0.8.1 → 0.8.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +29 -29
- package/TRADEMARKS.md +0 -1
- package/dist/src/config.d.ts +28 -0
- package/dist/src/config.js +20 -0
- package/dist/src/geometry.d.ts +21 -0
- package/dist/src/geometry.js +21 -0
- package/dist/src/meter.d.ts +76 -0
- package/dist/src/meter.js +95 -0
- package/dist/src/mind/attention.d.ts +4 -0
- package/dist/src/mind/attention.js +165 -16
- package/dist/src/mind/canonical.d.ts +16 -0
- package/dist/src/mind/canonical.js +41 -0
- package/dist/src/mind/corpus.d.ts +40 -0
- package/dist/src/mind/corpus.js +149 -0
- package/dist/src/mind/graph-search.d.ts +7 -0
- package/dist/src/mind/graph-search.js +254 -24
- package/dist/src/mind/index.d.ts +3 -1
- package/dist/src/mind/index.js +1 -0
- package/dist/src/mind/match.d.ts +9 -4
- package/dist/src/mind/match.js +147 -61
- package/dist/src/mind/mechanisms/cast.js +19 -3
- package/dist/src/mind/mechanisms/confluence.js +24 -0
- package/dist/src/mind/mechanisms/cover.js +6 -0
- package/dist/src/mind/mechanisms/recall.js +32 -4
- package/dist/src/mind/mind.d.ts +57 -0
- package/dist/src/mind/mind.js +72 -1
- package/dist/src/mind/pipeline-mechanism.d.ts +7 -0
- package/dist/src/mind/pipeline.js +66 -20
- package/dist/src/mind/primitives.js +9 -1
- package/dist/src/mind/rationale.d.ts +28 -1
- package/dist/src/mind/rationale.js +22 -1
- package/dist/src/mind/reasoning.d.ts +25 -3
- package/dist/src/mind/reasoning.js +125 -20
- package/dist/src/mind/recognition.js +4 -8
- package/dist/src/mind/resonance.js +20 -1
- package/dist/src/mind/trace.js +1 -0
- package/dist/src/mind/traverse.js +15 -3
- package/dist/src/mind/types.d.ts +49 -4
- package/docs/INVARIANTS.md +2 -2
- package/docs/architecture/bounded-reads.md +1 -1
- package/docs/architecture/commonality.md +2 -2
- package/docs/architecture/cost-model.md +2 -2
- package/docs/architecture/determinism.md +7 -7
- package/docs/architecture/match-project.md +2 -3
- package/docs/architecture/mechanism-market.md +10 -10
- package/docs/architecture/meter.md +5 -5
- package/docs/architecture/store.md +3 -3
- package/docs/failures/tempting-but-wrong.md +34 -6
- package/docs/harness/gates.md +2 -2
- package/docs/mechanisms/cast.md +2 -2
- package/docs/mechanisms/cover.md +2 -3
- package/docs/mechanisms/extraction.md +7 -7
- package/docs/mechanisms/recall.md +8 -9
- package/jsr.json +1 -1
- package/package.json +1 -1
- package/src/alu/README.md +11 -12
- package/src/config.ts +48 -0
- package/src/geometry.ts +21 -0
- package/src/meter.ts +98 -0
- package/src/mind/attention.ts +167 -16
- package/src/mind/canonical.ts +43 -0
- package/src/mind/corpus.ts +202 -0
- package/src/mind/graph-search.ts +277 -23
- package/src/mind/index.ts +8 -1
- package/src/mind/match.ts +148 -57
- package/src/mind/mechanisms/cast.ts +20 -2
- package/src/mind/mechanisms/confluence.ts +24 -0
- package/src/mind/mechanisms/cover.ts +5 -0
- package/src/mind/mechanisms/recall.ts +32 -4
- package/src/mind/mind.ts +125 -0
- package/src/mind/pipeline-mechanism.ts +7 -0
- package/src/mind/pipeline.ts +79 -22
- package/src/mind/primitives.ts +9 -1
- package/src/mind/rationale.ts +35 -1
- package/src/mind/reasoning.ts +145 -13
- package/src/mind/recognition.ts +4 -8
- package/src/mind/resonance.ts +19 -1
- package/src/mind/trace.ts +1 -0
- package/src/mind/traverse.ts +16 -6
- package/src/mind/types.ts +53 -4
- package/test/100-complete-grounding-trace.test.mjs +109 -0
- package/test/101-alignment-gap-bound.test.mjs +106 -0
- package/test/102-production-composes-at-scale.test.mjs +110 -0
- package/test/103-alignment-gap-budget.test.mjs +89 -0
- package/test/104-composition-is-reported.test.mjs +90 -0
- package/test/105-derive-through-reports-its-refusal.test.mjs +137 -0
- package/test/106-the-join-fires.test.mjs +94 -0
- package/test/107-the-join-is-counted.test.mjs +81 -0
- package/test/108-the-join-chains.test.mjs +78 -0
- package/test/109-the-pivot-is-counted.test.mjs +60 -0
- package/test/110-the-reasoner-stops-when-the-question-is-answered.test.mjs +91 -0
- package/test/111-the-cover-assembly-is-counted.test.mjs +74 -0
- package/test/112-the-exploration-does-not-grow-with-the-hub.test.mjs +89 -0
- package/test/113-the-rationale-payload-is-bounded.test.mjs +84 -0
- package/test/114-alignment-budget-is-per-sweep.test.mjs +93 -0
- package/test/116-the-extension-is-gated-by-the-pipelines-own-remainder.test.mjs +100 -0
- package/test/117-corpus-search.test.mjs +171 -0
- package/test/118-the-join-reaches-a-key-off-the-cut.test.mjs +74 -0
- package/test/119-the-work-does-not-grow-with-the-corpus.test.mjs +122 -0
- package/test/120-composition-is-consequence.test.mjs +132 -0
- package/test/121-the-extension-does-not-grow-with-the-corpus.test.mjs +128 -0
- package/test/122-the-climb-search-does-not-grow-with-the-corpus.test.mjs +117 -0
- package/test/123-the-paired-formulas-agree.test.mjs +90 -0
- package/test/125-the-post-grounding-branch-publishes-its-operand.test.mjs +51 -0
- package/test/126-the-pipeline-does-not-name-mechanisms.test.mjs +42 -0
- package/test/128-the-leads-somewhere-pair-agrees.test.mjs +83 -0
- package/test/129-the-trace-payload-shape.test.mjs +164 -0
- package/test/14-scaling.test.mjs +10 -7
- package/test/32-confluence.test.mjs +68 -0
- package/test/38-reason-restate-guard.test.mjs +8 -2
- package/test/43-cast-analog-seat.test.mjs +10 -0
- package/test/55-cost-meter.test.mjs +859 -0
- package/test/76-reference-binding.test.mjs +6 -1
- package/test/89-completion-recursion.test.mjs +30 -5
|
@@ -194,6 +194,13 @@ export class GraphSearch {
|
|
|
194
194
|
this.maxGroup = maxGroup;
|
|
195
195
|
this.host = host;
|
|
196
196
|
}
|
|
197
|
+
/** The nodes the QUERY canonically names — the same identity the store's keys
|
|
198
|
+
* were written through. A byte-exact test is not enough: the query writes
|
|
199
|
+
* `Eiffel Tower country` and the deposited node is `eiffel tower country`, so
|
|
200
|
+
* a join that filters the query's own subject by RAW bytes re-admits it —
|
|
201
|
+
* measured: that is the trap's wrong answer (`The capital of Eiffel Tower
|
|
202
|
+
* country is Berlin.`). Cached by query identity, because the search is
|
|
203
|
+
* reused across responses. */
|
|
197
204
|
/* * The hub bound √N (bounded-reads.md) — the ONE
|
|
198
205
|
* fan-out cap, stated here rather than imported from `traverse.ts` because
|
|
199
206
|
* this module is deliberately host-based (it holds a bare Store, never a
|
|
@@ -461,7 +468,7 @@ export class GraphSearch {
|
|
|
461
468
|
return this.coverRules(it, coversDone, coverableByStart);
|
|
462
469
|
}
|
|
463
470
|
if (it.kind === "form") {
|
|
464
|
-
return this.formRules(it, conceptTarget, substitutions, nodeBytes);
|
|
471
|
+
return this.formRules(it, conceptTarget, substitutions, nodeBytes, queryLen);
|
|
465
472
|
}
|
|
466
473
|
return this.outRules(it, {
|
|
467
474
|
W,
|
|
@@ -547,7 +554,7 @@ export class GraphSearch {
|
|
|
547
554
|
}
|
|
548
555
|
/** form(i,j,node,via): follow the graph out of `node`, or (in articulation)
|
|
549
556
|
* emit its substitute voice directly. */
|
|
550
|
-
*formRules(it, conceptTarget, substitutions, nodeBytes) {
|
|
557
|
+
*formRules(it, conceptTarget, substitutions, nodeBytes, queryLen) {
|
|
551
558
|
// Articulation: emit voice bytes at the recognised span; the hop/concept/
|
|
552
559
|
// emit chain is suppressed — the form contributes only its substitute.
|
|
553
560
|
if (substitutions) {
|
|
@@ -581,7 +588,46 @@ export class GraphSearch {
|
|
|
581
588
|
// guard then dead-ends it) with no way to reach the forward edge.
|
|
582
589
|
// Forking offers every continuation as its own rule so the one that
|
|
583
590
|
// genuinely advances (not a duplicate) is still reachable.
|
|
591
|
+
// A CHAIN HOP OFFERS ONLY WHAT THE QUESTION CAN PAY FOR.
|
|
592
|
+
//
|
|
593
|
+
// `hubBound` = √N is the READ cap — every read here stays inside it — but
|
|
594
|
+
// it is not an exploration bound: measured, a hub of degree 1083 sits
|
|
595
|
+
// BELOW √N = 1559, so a hop offered all 1083 continuations, the chart grew
|
|
596
|
+
// to 3113 outs for a two-word question, and since every out with an
|
|
597
|
+
// uncovered tail probes its tail's prefixes (measured: 16 885 canonical
|
|
598
|
+
// probes = 87% of that query's work, and its 270 MB peak / 256 MB OOM),
|
|
599
|
+
// the cost came from OFFERING rather than from reading.
|
|
600
|
+
//
|
|
601
|
+
// The bound is derived, not tuned: a derivation of L hops consumes ~L
|
|
602
|
+
// units of the question, so a hop cannot be paid for by offering more
|
|
603
|
+
// continuations than the question has units —
|
|
604
|
+
// `ceil(queryLen / W)`, floored at 2 for plurality. It is QUERY-sized
|
|
605
|
+
// (invariant 5: no per-query read grows with N) and it leaves `hubBound`
|
|
606
|
+
// and every read untouched.
|
|
607
|
+
// THE OFFER IS THE CORPUS'S OWN STRUCTURE, and the search pays for
|
|
608
|
+
// exploring it. There is no offer cap here any more: the traversal cap I
|
|
609
|
+
// had put on this hop was a short-circuit — it bounded what a hop could
|
|
610
|
+
// OFFER instead of charging for it — and it was not needed.
|
|
611
|
+
//
|
|
612
|
+
// MEASURED in the regime where it used to bite (`hubBound = ceil(√N)`
|
|
613
|
+
// GREATER than the hub's degree — reached in a fixture by choosing the
|
|
614
|
+
// degree below √N, so the trained store is not needed): with the cap the
|
|
615
|
+
// offer was 8/9/9 continuations at degrees 35/70/120; without it, 52/84/120
|
|
616
|
+
// — and the WORK is LINEAR in the degree, not quadratic: pushes 262/296/332,
|
|
617
|
+
// perceptions 530/592/757, while the PEAK is identical with and without the
|
|
618
|
+
// cap (218/415/689 MB against 215/410/662) because it is set by the store,
|
|
619
|
+
// not by the fan-out. What made this hop expensive was never the fan-out
|
|
620
|
+
// breadth: it was the per-offer work, two duplicate/oversized computations
|
|
621
|
+
// since removed (the per-offset canonical scan, and the tail scan now
|
|
622
|
+
// restricted to the fold's boundaries).
|
|
623
|
+
//
|
|
624
|
+
// The residual, stated: the trained store's hub (degree 1 083) is an
|
|
625
|
+
// EXTRAPOLATION from this linear shape, not a measurement.
|
|
584
626
|
const nx = this.store.nextFirst(it.node, this.hubBound());
|
|
627
|
+
// Count what is OFFERED, not what was read: the evidence-preferred
|
|
628
|
+
// continuation is yielded too, even when it lies outside the cap.
|
|
629
|
+
if (this.host.meter)
|
|
630
|
+
this.host.meter.chainOffers += nx.length + 1;
|
|
585
631
|
if (nx.length) {
|
|
586
632
|
// The SAME evidence-weighted disambiguation the first hop uses
|
|
587
633
|
// (below) identifies the most-corroborated continuation. Yielding
|
|
@@ -590,10 +636,14 @@ export class GraphSearch {
|
|
|
590
636
|
// arrivals at an EQUAL cost (`cost < current`, strictly) — so
|
|
591
637
|
// among same-depth sibling forks that tie in cost, the
|
|
592
638
|
// evidence-backed edge wins deterministically, never by
|
|
593
|
-
// exploration-order luck. `preferred
|
|
594
|
-
//
|
|
595
|
-
// set
|
|
596
|
-
//
|
|
639
|
+
// exploration-order luck. `preferred` is NOT necessarily an element
|
|
640
|
+
// of `nx` any more: `nx` is the capped read above, while `chooseNext`
|
|
641
|
+
// reads its own hub-bounded set (see traverse.ts) — so the evidence
|
|
642
|
+
// pick may lie outside the cap, and it is still yielded FIRST on
|
|
643
|
+
// purpose. The cap bounds what the hop EXPLORES; it must never make
|
|
644
|
+
// the evidence-ranked continuation unreachable, or the derivation the
|
|
645
|
+
// corpus corroborates would lose to exploration order. Nothing is
|
|
646
|
+
// reallocated: `nx` is walked with a skip-in-place for the duplicate.
|
|
597
647
|
const preferred = nx.length > 1
|
|
598
648
|
? this.host.chooseNext?.(it.node)
|
|
599
649
|
: undefined;
|
|
@@ -818,10 +868,50 @@ export class GraphSearch {
|
|
|
818
868
|
// concepts/connectors either (those need the caller's async
|
|
819
869
|
// pre-resolution) — the recursion follows edges and fusion, which is what
|
|
820
870
|
// a deeper rewrite chain is made of.
|
|
871
|
+
if (this.host.meter)
|
|
872
|
+
this.host.meter.recompletes++;
|
|
821
873
|
const rec = this.host.recogniseSpan(bytes);
|
|
822
874
|
const kids = new Set(nrec.kids);
|
|
875
|
+
// THE NODE'S OWN KIDS ARE SITES BY STRUCTURE — recognition cannot be the
|
|
876
|
+
// only source. A produced composite is decomposed by ITS OWN SHAPE, and
|
|
877
|
+
// at hub scale the recognition of a produced span returns the WHOLE while
|
|
878
|
+
// deliberately suppressing its atoms (the off-boundary suppression), so
|
|
879
|
+
// the kid filter below would admit nothing at all and the chain would end
|
|
880
|
+
// at the intermediate composite. Measured on the chain
|
|
881
|
+
// `seed → "p q" → (p→r, q→s) → "r s" → "m n"`: below the flip it reaches
|
|
882
|
+
// "m n" with fuse+recompose, above it stops at "p q" — the trace shows
|
|
883
|
+
// `recognise("p q") ⇒ form "p q"` alone, no parts.
|
|
884
|
+
//
|
|
885
|
+
// Laying the node's kids out over its own bytes restores exactly the
|
|
886
|
+
// decomposition the node's tree already states; a kid recognition ALREADY
|
|
887
|
+
// offers is skipped, so below the flip the seed set is byte-identical to
|
|
888
|
+
// what it was. The filter's guarantee is untouched: nothing beyond the
|
|
889
|
+
// node's own kids may enter.
|
|
890
|
+
const recognised = rec.sites.filter((s) => kids.has(s.payload));
|
|
891
|
+
// DEDUPED BY SPAN, not by payload. A node whose kids repeat (`"abab"`
|
|
892
|
+
// folds as ["ab","ab"]) has TWO occurrences of the same node at different
|
|
893
|
+
// offsets; a payload-keyed set suppressed the structural site for BOTH, so
|
|
894
|
+
// the second occurrence had no site at all. The set now holds the spans
|
|
895
|
+
// recognition already covers, and a structural site is added exactly when
|
|
896
|
+
// nothing covers that PLACE — O(1) lookups, no extra reads.
|
|
897
|
+
const seenSpans = new Set(recognised.map((s) => `${s.start}:${s.end}`));
|
|
898
|
+
const structural = [];
|
|
899
|
+
{
|
|
900
|
+
let off = 0;
|
|
901
|
+
for (const k of nrec.kids) {
|
|
902
|
+
const len = this.store.bytesPrefix(k, ALL).length;
|
|
903
|
+
if (!seenSpans.has(`${off}:${off + len}`) && len > 0) {
|
|
904
|
+
structural.push({
|
|
905
|
+
start: off,
|
|
906
|
+
end: Math.min(off + len, bytes.length),
|
|
907
|
+
payload: k,
|
|
908
|
+
});
|
|
909
|
+
}
|
|
910
|
+
off += len;
|
|
911
|
+
}
|
|
912
|
+
}
|
|
823
913
|
const solved = this.solve(bytes.length, {
|
|
824
|
-
sites:
|
|
914
|
+
sites: [...recognised, ...structural],
|
|
825
915
|
leaves: rec.leaves,
|
|
826
916
|
splits: rec.splits,
|
|
827
917
|
starts: rec.starts,
|
|
@@ -892,6 +982,14 @@ export class GraphSearch {
|
|
|
892
982
|
const tail = queryBytes.subarray(fact.j, queryLen);
|
|
893
983
|
if (tail.length === 0)
|
|
894
984
|
return;
|
|
985
|
+
// Report ONLY the invocations that could have joined: the search asks this
|
|
986
|
+
// rule for every finalized out with a node, which includes the one-byte
|
|
987
|
+
// outs the cover bridges with — measured, 68 refusals for a single
|
|
988
|
+
// 3-relation query, all of them letters. A form shorter than one window is
|
|
989
|
+
// not a fact a join could travel through, so it is not a refusal worth
|
|
990
|
+
// reporting; W is the same line the rest of the mind draws between a chance
|
|
991
|
+
// overlap and a form.
|
|
992
|
+
const reportable = fact.bytes.length >= this.maxGroup;
|
|
895
993
|
// The entity candidates are the forms the fact's own bytes CONTAIN — the
|
|
896
994
|
// same recogniser the query went through, so the evidence standard is the
|
|
897
995
|
// query's. A byte atom is never a subject; the fact's own node is the span
|
|
@@ -900,10 +998,62 @@ export class GraphSearch {
|
|
|
900
998
|
// LENDS it when it can (Mind does, with the response-scoped struct cache),
|
|
901
999
|
// and a bare host falls back to the raw-store probe, so the search stays
|
|
902
1000
|
// host-based.
|
|
903
|
-
const
|
|
904
|
-
|
|
905
|
-
|
|
906
|
-
|
|
1001
|
+
const factRec = this.host.recogniseSpan(fact.bytes);
|
|
1002
|
+
const leads = (id) => this.host.leadsSomewhere !== undefined
|
|
1003
|
+
? this.host.leadsSomewhere(id)
|
|
1004
|
+
: this.store.hasNext(id) || this.store.hasHalo(id);
|
|
1005
|
+
// THE QUERY'S OWN SUBJECT, CANONICALLY. The filter used raw bytes and the
|
|
1006
|
+
// store's nodes are canonical, so `Eiffel Tower country` in the query did
|
|
1007
|
+
// not match the deposited `eiffel tower country` — measured, that is the
|
|
1008
|
+
// trap's wrong answer.
|
|
1009
|
+
//
|
|
1010
|
+
// TAKEN FROM THE RECOGNITION THE RESPONSE ALREADY COMPUTED — the host's
|
|
1011
|
+
// `recogniseSpan`, the same surface the rest of this search uses — not from
|
|
1012
|
+
// a second offset scan. `canonicalQueryNodes` re-derived, per byte offset,
|
|
1013
|
+
// what `recognise` had already resolved once per query (its memo is keyed by
|
|
1014
|
+
// content), and that scan was the largest single cost the DIANOT join added:
|
|
1015
|
+
// measured against the pre-change tree, the same fixture and the same test
|
|
1016
|
+
// were 21 s slower with the scan than without it. A recognised site IS a
|
|
1017
|
+
// canonical node of the query that can lead somewhere, which is exactly the
|
|
1018
|
+
// set this filter wants, and it costs nothing to read.
|
|
1019
|
+
const queryNodes = new Set((this.host.recogniseSpan?.(queryBytes)?.sites ?? []).map((s) => s.payload));
|
|
1020
|
+
// TWO SOURCES, ONE ADMISSION. The recognition of a STORED WHOLE returns the
|
|
1021
|
+
// whole and stops — measured: for `The director of Eva is Gustaf Molander.`
|
|
1022
|
+
// it yields exactly ONE site, the fact's own node — so the entity a join
|
|
1023
|
+
// exists for is never proposed. The canonical fold is the second source,
|
|
1024
|
+
// and the scan runs only for a FORM (≥ W: a one-byte out is not something to
|
|
1025
|
+
// join through, and running it per letter measured 20-26 s in test/99).
|
|
1026
|
+
const W = this.maxGroup;
|
|
1027
|
+
const proposed = new Map();
|
|
1028
|
+
// The SOURCE of each proposal travels with it: a refusal that names only the
|
|
1029
|
+
// bytes leaves the next reader guessing which path proposed them — three
|
|
1030
|
+
// attempts at the chained join were spent fixing paths that never produced
|
|
1031
|
+
// the offending candidate.
|
|
1032
|
+
const source = new Map();
|
|
1033
|
+
for (const s of factRec.sites) {
|
|
1034
|
+
if (s.payload >= 0 && leads(s.payload)) {
|
|
1035
|
+
proposed.set(s.payload, this.store.bytesPrefix(s.payload, ALL));
|
|
1036
|
+
source.set(s.payload, "recognised site");
|
|
1037
|
+
}
|
|
1038
|
+
}
|
|
1039
|
+
if (this.host.canonResolve !== undefined && fact.bytes.length >= W) {
|
|
1040
|
+
const canon = this.host.canonResolve.bind(this.host);
|
|
1041
|
+
for (let start = 0; start < fact.bytes.length; start++) {
|
|
1042
|
+
for (let end = fact.bytes.length; end - start >= W; end--) {
|
|
1043
|
+
const id = canon(fact.bytes.subarray(start, end));
|
|
1044
|
+
if (id === null)
|
|
1045
|
+
continue;
|
|
1046
|
+
if (leads(id)) {
|
|
1047
|
+
proposed.set(id, this.store.bytesPrefix(id, ALL));
|
|
1048
|
+
source.set(id, "canonical fold");
|
|
1049
|
+
}
|
|
1050
|
+
break; // the longest form at this offset wins
|
|
1051
|
+
}
|
|
1052
|
+
}
|
|
1053
|
+
}
|
|
1054
|
+
const leading = [...proposed]
|
|
1055
|
+
.filter(([payload]) => payload !== fact.node && !queryNodes.has(payload))
|
|
1056
|
+
.map(([payload, bytes]) => ({ payload, bytes }));
|
|
907
1057
|
// …then prefer the entity the query did NOT name, and the MAXIMAL one. The
|
|
908
1058
|
// join exists to reach the subject the query never wrote, so:
|
|
909
1059
|
// • a candidate the query already contains is the query's OWN subject, and
|
|
@@ -914,30 +1064,110 @@ export class GraphSearch {
|
|
|
914
1064
|
// introduces ("Timur" must not win over "Timur Bekmambetov").
|
|
915
1065
|
// Byte work over bytes already read, and the pruning REMOVES the
|
|
916
1066
|
// resolve()/nextFirst() probes these candidates would have paid.
|
|
1067
|
+
if (leading.length === 0) {
|
|
1068
|
+
if (this.host.meter)
|
|
1069
|
+
this.host.meter.joinNoEntity++;
|
|
1070
|
+
if (reportable) {
|
|
1071
|
+
// Report WHAT the recognition returned, not just that nothing led: the
|
|
1072
|
+
// count and the first few site texts are the difference between "the
|
|
1073
|
+
// fact was not recognised" and "it was recognised but nothing led".
|
|
1074
|
+
const seen = factRec.sites.slice(0, 3).map((s) => this.store.bytesPrefix(s.payload, ALL));
|
|
1075
|
+
this.host.reportSearch?.("deriveThroughMiss", [fact.bytes, tail, ...seen], `no entity inside the fact leads anywhere — ${factRec.sites.length} site(s) recognised inside it`);
|
|
1076
|
+
}
|
|
1077
|
+
}
|
|
917
1078
|
const candidates = leading
|
|
918
|
-
.map((s) => ({
|
|
919
|
-
payload: s.payload,
|
|
920
|
-
bytes: this.store.bytesPrefix(s.payload, ALL),
|
|
921
|
-
}))
|
|
922
|
-
.filter((c) => indexOf(queryBytes, c.bytes, 0) < 0)
|
|
923
1079
|
.filter((c, _i, all) => !all.some((o) => o.bytes.length > c.bytes.length && indexOf(o.bytes, c.bytes, 0) >= 0));
|
|
924
1080
|
for (const c of candidates) {
|
|
925
|
-
|
|
926
|
-
|
|
927
|
-
|
|
928
|
-
|
|
929
|
-
|
|
1081
|
+
// THE SHORTEST TAIL PREFIX WHOSE KEY ALSO LEADS SOMEWHERE.
|
|
1082
|
+
//
|
|
1083
|
+
// The conclusion covers only that prefix, so the rest of the tail stays
|
|
1084
|
+
// for the step after — which is what CHAINING is (a whole-tail key
|
|
1085
|
+
// consumed the whole remainder and made every join terminal: measured,
|
|
1086
|
+
// `joinFired=0` on a three-relation query).
|
|
1087
|
+
//
|
|
1088
|
+
// A KEY THAT RESOLVES IS NOT ENOUGH. Measured with a dry run of these
|
|
1089
|
+
// very primitives: for the candidate `Sweden` the first tail prefix that
|
|
1090
|
+
// resolves is `" "` — the key `Sweden ` (a trailing space) — and it leads
|
|
1091
|
+
// NOWHERE (`nextFirst` = 0), so accepting it refused the join while the
|
|
1092
|
+
// key that names the fact (`sweden capital`) sat one prefix further. The
|
|
1093
|
+
// loop therefore asks BOTH questions before accepting, and keeps looking
|
|
1094
|
+
// otherwise.
|
|
1095
|
+
//
|
|
1096
|
+
// EXACT FIRST, THEN CANONICAL: the corpus holds both identities.
|
|
1097
|
+
let key = null;
|
|
1098
|
+
let used = 0;
|
|
1099
|
+
// The continuation the accepted key leads to, carried out of the loop:
|
|
1100
|
+
// the loop already HAD to read it to accept the key (a key that leads
|
|
1101
|
+
// nowhere is not the relation), so re-reading it after the loop was a
|
|
1102
|
+
// duplicate read and a branch that could never be taken.
|
|
1103
|
+
let next = null;
|
|
1104
|
+
let keyBytes = c.bytes;
|
|
1105
|
+
// THE CANDIDATE ENDS ARE THE PREFIXES THAT ARE STORED NODES, ASCENDING.
|
|
1106
|
+
// The key is `entity + prefix`, and it names a relation exactly when that
|
|
1107
|
+
// concatenation IS a node — so the ends come from a content-addressed
|
|
1108
|
+
// probe per offset (the host's `contentKeyEnds`, the learning path's own
|
|
1109
|
+
// mechanism: one leaf walk plus one `findBranch` per offset, no `resolve`),
|
|
1110
|
+
// never from the fold's boundaries. A boundary is not a proxy: a stored
|
|
1111
|
+
// member's end is the end of ITS OWN stream, and the fold never cuts at a
|
|
1112
|
+
// stream's end — measured, "stockholm mayor" exists, leads on to the mayor
|
|
1113
|
+
// fact, and its boundary 6 sits in neither the tail's cuts ([4,7]) nor the
|
|
1114
|
+
// concatenation's. Filtering the scan by "is this a node?" cannot change
|
|
1115
|
+
// the winner: a position that is not a node cannot resolve, so skipping it
|
|
1116
|
+
// is invisible; and the order stays SHORTEST FIRST, which is a semantic
|
|
1117
|
+
// law, not an optimisation (test/106, test/108 pin it).
|
|
1118
|
+
//
|
|
1119
|
+
// A host that cannot answer falls back to every prefix: exact and
|
|
1120
|
+
// complete, at a `resolve` per offset. A host that CAN answer is
|
|
1121
|
+
// authoritative even when it answers "none" — if no prefix is a node then
|
|
1122
|
+
// no key exists to resolve, so enumerating would only pay nulls. (A key
|
|
1123
|
+
// reachable through the CANONICAL equivalence alone and ending off every
|
|
1124
|
+
// node end is therefore not tried here; that dimension is unreachable on
|
|
1125
|
+
// this path by construction and is not part of the exact-key law.)
|
|
1126
|
+
const ends = this.host.contentKeyEnds?.(c.bytes, tail);
|
|
1127
|
+
const candidateEnds = function* () {
|
|
1128
|
+
if (ends !== undefined) {
|
|
1129
|
+
yield* ends;
|
|
1130
|
+
return;
|
|
1131
|
+
}
|
|
1132
|
+
for (let p = 1; p <= tail.length; p++)
|
|
1133
|
+
yield p;
|
|
1134
|
+
};
|
|
1135
|
+
for (const len of candidateEnds()) {
|
|
1136
|
+
keyBytes = concat2(c.bytes, tail.subarray(0, len));
|
|
1137
|
+
const k = this.host.resolve(keyBytes) ??
|
|
1138
|
+
this.host.canonResolve?.(keyBytes) ??
|
|
1139
|
+
null;
|
|
1140
|
+
if (k === null)
|
|
1141
|
+
continue;
|
|
1142
|
+
const nx = this.store.nextFirst(k, 1);
|
|
1143
|
+
if (nx.length === 0)
|
|
1144
|
+
continue;
|
|
1145
|
+
key = k;
|
|
1146
|
+
used = len;
|
|
1147
|
+
next = nx[0];
|
|
1148
|
+
break;
|
|
1149
|
+
}
|
|
1150
|
+
if (key === null) {
|
|
1151
|
+
if (this.host.meter)
|
|
1152
|
+
this.host.meter.joinNoKey++;
|
|
1153
|
+
if (reportable) {
|
|
1154
|
+
this.host.reportSearch?.("deriveThroughMiss", [c.bytes, tail, keyBytes], `no learnt key names this entity and tail together ` +
|
|
1155
|
+
`(candidate #${c.payload}, from the ${source.get(c.payload) ?? "unknown"} source)`);
|
|
1156
|
+
}
|
|
930
1157
|
continue;
|
|
1158
|
+
}
|
|
1159
|
+
if (this.host.meter)
|
|
1160
|
+
this.host.meter.joinFired++;
|
|
931
1161
|
yield {
|
|
932
1162
|
premises: [fact],
|
|
933
1163
|
conclusion: {
|
|
934
1164
|
kind: "out",
|
|
935
1165
|
i: fact.i,
|
|
936
|
-
j:
|
|
937
|
-
bytes: this.store.bytesPrefix(
|
|
1166
|
+
j: fact.j + used,
|
|
1167
|
+
bytes: this.store.bytesPrefix(next, ALL),
|
|
938
1168
|
cover: true,
|
|
939
1169
|
rec: true,
|
|
940
|
-
node:
|
|
1170
|
+
node: next,
|
|
941
1171
|
throughFact: true,
|
|
942
1172
|
},
|
|
943
1173
|
cost: STEP,
|
package/dist/src/mind/index.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
export { Mind } from "./mind.js";
|
|
2
|
-
export type { Input, Response } from "./mind.js";
|
|
2
|
+
export type { CorpusTextPair, CorpusTextResult, Input, Response, } from "./mind.js";
|
|
3
3
|
export type { ComputedSpan, ExtensionHost } from "./mind.js";
|
|
4
4
|
export type { MechanismResult, PipelineMechanism, Precomputed, } from "./pipeline-mechanism.js";
|
|
5
5
|
export type { InspectRationale, RationaleItem, RationaleStep, } from "./rationale.js";
|
|
@@ -7,3 +7,5 @@ export type { AnchorRejectionReason, ClimbConsensusData, ConsensusAnchorTrace, C
|
|
|
7
7
|
export type { AncestorReach, AttentionRead, SaturationReason, SaturationStop, } from "./types.js";
|
|
8
8
|
export type { DepositReport } from "./learning.js";
|
|
9
9
|
export type { DecideGroundingData, NarrowDecisionData, Provenance, } from "./pipeline.js";
|
|
10
|
+
export { sampleCorpus, searchCorpus } from "./corpus.js";
|
|
11
|
+
export type { CorpusMiss, CorpusPair, CorpusResult } from "./corpus.js";
|
package/dist/src/mind/index.js
CHANGED
package/dist/src/mind/match.d.ts
CHANGED
|
@@ -78,9 +78,14 @@ export interface AlignGap {
|
|
|
78
78
|
}
|
|
79
79
|
/** Extend a seed match (query offset qo ↔ candidate offset co) to its maximal
|
|
80
80
|
* common run, then walk outward in both directions collecting further common
|
|
81
|
-
* runs of at least W bytes across
|
|
82
|
-
*
|
|
83
|
-
*
|
|
81
|
+
* runs of at least W bytes across mismatch gaps. Each gap's LENGTH is the
|
|
82
|
+
* pair's own extent (a gap cannot be longer than the bytes it spans) and the
|
|
83
|
+
* sweep's WORK is proportional to the bytes a run spans (the context's windows
|
|
84
|
+
* are indexed once, then the query's are walked) — the arity bound
|
|
85
|
+
* (`chainReach`) used to cap BOTH, and truncated every learned frame whose
|
|
86
|
+
* slot was longer. Each sweep owns its own budget, so an exhausted right
|
|
87
|
+
* sweep never starves the left one. Returns the matched query spans and the
|
|
88
|
+
* mismatch pairs between consecutive runs.
|
|
84
89
|
*
|
|
85
90
|
* This is the SEEDED aligner, distinct from {@link alignRuns}: that one finds
|
|
86
91
|
* every run two structures share anywhere (a weave), this one reads two
|
|
@@ -195,7 +200,7 @@ export declare function frameSlots(ctx: MindContext, query: Uint8Array, cand: Ui
|
|
|
195
200
|
* ANCHOR that the query displaced. Neither implies the other, and the
|
|
196
201
|
* observed failures pass the restatement guard cleanly.
|
|
197
202
|
*
|
|
198
|
-
*
|
|
203
|
+
* Four conditions, all byte-exact and all necessary:
|
|
199
204
|
*
|
|
200
205
|
* 1. the query and the anchor must be ONE STRUCTURE — what they share has to
|
|
201
206
|
* dominate the query, or the query is not a variant of the anchor at all
|