@hviana/sema 0.8.0 → 0.8.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +22 -1
- package/DATASETS.md +1 -1
- package/dist/example/train_base/config.js +2 -2
- package/dist/example/train_base/corpora/massive.js +1 -1
- package/dist/example/train_base/readers.js +1 -1
- package/dist/src/config.d.ts +17 -0
- package/dist/src/config.js +18 -0
- package/dist/src/geometry.d.ts +10 -10
- package/dist/src/geometry.js +25 -24
- package/dist/src/meter.d.ts +29 -12
- package/dist/src/meter.js +58 -14
- package/dist/src/mind/attention.js +12 -12
- package/dist/src/mind/bridge.d.ts +8 -8
- package/dist/src/mind/bridge.js +33 -32
- package/dist/src/mind/corpus.d.ts +40 -0
- package/dist/src/mind/corpus.js +149 -0
- package/dist/src/mind/graph-search.d.ts +7 -8
- package/dist/src/mind/graph-search.js +244 -32
- package/dist/src/mind/index.d.ts +3 -1
- package/dist/src/mind/index.js +1 -0
- package/dist/src/mind/junction.d.ts +1 -1
- package/dist/src/mind/junction.js +8 -8
- package/dist/src/mind/learning.js +36 -35
- package/dist/src/mind/match.d.ts +8 -3
- package/dist/src/mind/match.js +156 -71
- package/dist/src/mind/mechanisms/cast.js +18 -2
- package/dist/src/mind/mechanisms/cover.js +19 -12
- package/dist/src/mind/mechanisms/prefix-completion.js +24 -24
- package/dist/src/mind/mechanisms/recall.js +38 -40
- package/dist/src/mind/mechanisms/reference.js +16 -16
- package/dist/src/mind/mind.d.ts +61 -7
- package/dist/src/mind/mind.js +72 -2
- package/dist/src/mind/pipeline-mechanism.d.ts +10 -8
- package/dist/src/mind/pipeline-mechanism.js +25 -21
- package/dist/src/mind/pipeline.d.ts +9 -9
- package/dist/src/mind/pipeline.js +49 -29
- package/dist/src/mind/primitives.d.ts +5 -5
- package/dist/src/mind/primitives.js +5 -5
- package/dist/src/mind/reasoning.d.ts +5 -1
- package/dist/src/mind/reasoning.js +54 -1
- package/dist/src/mind/recognition.d.ts +14 -13
- package/dist/src/mind/recognition.js +23 -23
- package/dist/src/mind/resonance.js +21 -21
- package/dist/src/mind/traverse.d.ts +54 -52
- package/dist/src/mind/traverse.js +83 -73
- package/dist/src/mind/types.d.ts +26 -4
- package/dist/src/store.d.ts +12 -12
- package/dist/src/store.js +12 -12
- package/docs/INDEX.md +2 -2
- package/docs/architecture/exact-vs-approximate.md +2 -1
- package/docs/architecture/fold-contract.md +1 -1
- package/docs/failures/tempting-but-wrong.md +33 -5
- package/docs/harness/gates.md +7 -7
- package/example/train_base/config.ts +2 -2
- package/example/train_base/corpora/massive.ts +1 -1
- package/example/train_base/readers.ts +1 -1
- package/jsr.json +1 -1
- package/package.json +1 -1
- package/src/config.ts +35 -0
- package/src/geometry.ts +25 -24
- package/src/meter.ts +61 -14
- package/src/mind/attention.ts +12 -12
- package/src/mind/bridge.ts +33 -32
- package/src/mind/corpus.ts +202 -0
- package/src/mind/graph-search.ts +261 -31
- package/src/mind/index.ts +8 -1
- package/src/mind/junction.ts +8 -8
- package/src/mind/learning.ts +36 -35
- package/src/mind/match.ts +163 -73
- package/src/mind/mechanisms/cast.ts +17 -1
- package/src/mind/mechanisms/cover.ts +18 -12
- package/src/mind/mechanisms/prefix-completion.ts +24 -24
- package/src/mind/mechanisms/recall.ts +38 -40
- package/src/mind/mechanisms/reference.ts +16 -16
- package/src/mind/mind.ts +129 -7
- package/src/mind/pipeline-mechanism.ts +25 -21
- package/src/mind/pipeline.ts +63 -38
- package/src/mind/primitives.ts +5 -5
- package/src/mind/reasoning.ts +55 -0
- package/src/mind/recognition.ts +23 -23
- package/src/mind/resonance.ts +21 -21
- package/src/mind/traverse.ts +83 -73
- package/src/mind/types.ts +30 -4
- package/src/store.ts +20 -20
- package/test/08-storage.test.mjs +1 -1
- package/test/100-complete-grounding-trace.test.mjs +109 -0
- package/test/101-alignment-gap-bound.test.mjs +106 -0
- package/test/102-production-composes-at-scale.test.mjs +110 -0
- package/test/103-alignment-gap-budget.test.mjs +89 -0
- package/test/104-composition-is-reported.test.mjs +90 -0
- package/test/105-derive-through-reports-its-refusal.test.mjs +113 -0
- package/test/106-the-join-fires.test.mjs +94 -0
- package/test/107-the-join-is-counted.test.mjs +81 -0
- package/test/108-the-join-chains.test.mjs +78 -0
- package/test/109-the-pivot-is-counted.test.mjs +60 -0
- package/test/110-the-reasoner-stops-when-the-question-is-answered.test.mjs +91 -0
- package/test/111-the-cover-assembly-is-counted.test.mjs +74 -0
- package/test/112-the-exploration-does-not-grow-with-the-hub.test.mjs +89 -0
- package/test/113-the-rationale-payload-is-bounded.test.mjs +84 -0
- package/test/114-alignment-budget-is-per-sweep.test.mjs +93 -0
- package/test/116-the-extension-is-gated-by-the-pipelines-own-remainder.test.mjs +100 -0
- package/test/117-corpus-search.test.mjs +171 -0
- package/test/14-scaling.test.mjs +10 -7
- package/test/35-prefix-edge.test.mjs +1 -1
- package/test/40-choosenext-scale-guard.test.mjs +16 -17
- package/test/56-bridge-identity-admission.test.mjs +6 -6
- package/test/70-prefix-completion.test.mjs +4 -3
- package/test/72-prefix-candidate-supply.test.mjs +3 -3
- package/test/73-scaffolding-only-bridge-abstains.test.mjs +6 -6
- package/test/75-multiturn-context-optimisation.test.mjs +5 -5
- package/test/76-reference-binding.test.mjs +6 -1
- package/test/84-composed-answer-honesty.test.mjs +5 -6
- package/test/88-dependency-footprint.test.mjs +1 -1
- package/test/89-completion-recursion.test.mjs +47 -19
- package/test/90-connector-read-cap.test.mjs +10 -8
- package/test/93-regime-prediction.test.mjs +10 -10
- package/test/94-cross-region-budget.test.mjs +2 -2
- package/test/95-wide-resonance-removed.test.mjs +8 -7
- package/test/96-bytes-walk-termination.test.mjs +3 -3
package/src/mind/graph-search.ts
CHANGED
|
@@ -375,9 +375,18 @@ export class GraphSearch {
|
|
|
375
375
|
private readonly host: GraphSearchHost,
|
|
376
376
|
) {}
|
|
377
377
|
|
|
378
|
-
/** The
|
|
379
|
-
*
|
|
380
|
-
*
|
|
378
|
+
/** The nodes the QUERY canonically names — the same identity the store's keys
|
|
379
|
+
* were written through. A byte-exact test is not enough: the query writes
|
|
380
|
+
* `Eiffel Tower country` and the deposited node is `eiffel tower country`, so
|
|
381
|
+
* a join that filters the query's own subject by RAW bytes re-admits it —
|
|
382
|
+
* measured: that is the trap's wrong answer (`The capital of Eiffel Tower
|
|
383
|
+
* country is Berlin.`). Cached by query identity, because the search is
|
|
384
|
+
* reused across responses. */
|
|
385
|
+
|
|
386
|
+
/* * The hub bound √N (bounded-reads.md) — the ONE
|
|
387
|
+
* fan-out cap, stated here rather than imported from `traverse.ts` because
|
|
388
|
+
* this module is deliberately host-based (it holds a bare Store, never a
|
|
389
|
+
* MindContext).
|
|
381
390
|
* That is the same write/read-side duplication convention canonical.ts's
|
|
382
391
|
* header documents: if the formula changes it must change in BOTH places.
|
|
383
392
|
* It is stated ONCE per side, though — the expression used to be spelled
|
|
@@ -702,7 +711,13 @@ export class GraphSearch {
|
|
|
702
711
|
return this.coverRules(it, coversDone, coverableByStart);
|
|
703
712
|
}
|
|
704
713
|
if (it.kind === "form") {
|
|
705
|
-
return this.formRules(
|
|
714
|
+
return this.formRules(
|
|
715
|
+
it,
|
|
716
|
+
conceptTarget,
|
|
717
|
+
substitutions,
|
|
718
|
+
nodeBytes,
|
|
719
|
+
queryLen,
|
|
720
|
+
);
|
|
706
721
|
}
|
|
707
722
|
return this.outRules(it, {
|
|
708
723
|
W,
|
|
@@ -807,6 +822,7 @@ export class GraphSearch {
|
|
|
807
822
|
conceptTarget: ReadonlyMap<number, number>,
|
|
808
823
|
substitutions: ReadonlyMap<number, Uint8Array> | undefined,
|
|
809
824
|
nodeBytes: (n: number) => Uint8Array,
|
|
825
|
+
queryLen: number,
|
|
810
826
|
): Iterable<Rule<GItem>> {
|
|
811
827
|
// Articulation: emit voice bytes at the recognised span; the hop/concept/
|
|
812
828
|
// emit chain is suppressed — the form contributes only its substitute.
|
|
@@ -842,7 +858,45 @@ export class GraphSearch {
|
|
|
842
858
|
// guard then dead-ends it) with no way to reach the forward edge.
|
|
843
859
|
// Forking offers every continuation as its own rule so the one that
|
|
844
860
|
// genuinely advances (not a duplicate) is still reachable.
|
|
861
|
+
// A CHAIN HOP OFFERS ONLY WHAT THE QUESTION CAN PAY FOR.
|
|
862
|
+
//
|
|
863
|
+
// `hubBound` = √N is the READ cap — every read here stays inside it — but
|
|
864
|
+
// it is not an exploration bound: measured, a hub of degree 1083 sits
|
|
865
|
+
// BELOW √N = 1559, so a hop offered all 1083 continuations, the chart grew
|
|
866
|
+
// to 3113 outs for a two-word question, and since every out with an
|
|
867
|
+
// uncovered tail probes its tail's prefixes (measured: 16 885 canonical
|
|
868
|
+
// probes = 87% of that query's work, and its 270 MB peak / 256 MB OOM),
|
|
869
|
+
// the cost came from OFFERING rather than from reading.
|
|
870
|
+
//
|
|
871
|
+
// The bound is derived, not tuned: a derivation of L hops consumes ~L
|
|
872
|
+
// units of the question, so a hop cannot be paid for by offering more
|
|
873
|
+
// continuations than the question has units —
|
|
874
|
+
// `ceil(queryLen / W)`, floored at 2 for plurality. It is QUERY-sized
|
|
875
|
+
// (invariant 5: no per-query read grows with N) and it leaves `hubBound`
|
|
876
|
+
// and every read untouched.
|
|
877
|
+
// THE OFFER IS THE CORPUS'S OWN STRUCTURE, and the search pays for
|
|
878
|
+
// exploring it. There is no offer cap here any more: the traversal cap I
|
|
879
|
+
// had put on this hop was a short-circuit — it bounded what a hop could
|
|
880
|
+
// OFFER instead of charging for it — and it was not needed.
|
|
881
|
+
//
|
|
882
|
+
// MEASURED in the regime where it used to bite (`hubBound = ceil(√N)`
|
|
883
|
+
// GREATER than the hub's degree — reached in a fixture by choosing the
|
|
884
|
+
// degree below √N, so the trained store is not needed): with the cap the
|
|
885
|
+
// offer was 8/9/9 continuations at degrees 35/70/120; without it, 52/84/120
|
|
886
|
+
// — and the WORK is LINEAR in the degree, not quadratic: pushes 262/296/332,
|
|
887
|
+
// perceptions 530/592/757, while the PEAK is identical with and without the
|
|
888
|
+
// cap (218/415/689 MB against 215/410/662) because it is set by the store,
|
|
889
|
+
// not by the fan-out. What made this hop expensive was never the fan-out
|
|
890
|
+
// breadth: it was the per-offer work, two duplicate/oversized computations
|
|
891
|
+
// since removed (the per-offset canonical scan, and the tail scan now
|
|
892
|
+
// restricted to the fold's boundaries).
|
|
893
|
+
//
|
|
894
|
+
// The residual, stated: the trained store's hub (degree 1 083) is an
|
|
895
|
+
// EXTRAPOLATION from this linear shape, not a measurement.
|
|
845
896
|
const nx = this.store.nextFirst(it.node, this.hubBound());
|
|
897
|
+
// Count what is OFFERED, not what was read: the evidence-preferred
|
|
898
|
+
// continuation is yielded too, even when it lies outside the cap.
|
|
899
|
+
if (this.host.meter) this.host.meter.chainOffers += nx.length + 1;
|
|
846
900
|
if (nx.length) {
|
|
847
901
|
// The SAME evidence-weighted disambiguation the first hop uses
|
|
848
902
|
// (below) identifies the most-corroborated continuation. Yielding
|
|
@@ -851,10 +905,14 @@ export class GraphSearch {
|
|
|
851
905
|
// arrivals at an EQUAL cost (`cost < current`, strictly) — so
|
|
852
906
|
// among same-depth sibling forks that tie in cost, the
|
|
853
907
|
// evidence-backed edge wins deterministically, never by
|
|
854
|
-
// exploration-order luck. `preferred
|
|
855
|
-
//
|
|
856
|
-
// set
|
|
857
|
-
//
|
|
908
|
+
// exploration-order luck. `preferred` is NOT necessarily an element
|
|
909
|
+
// of `nx` any more: `nx` is the capped read above, while `chooseNext`
|
|
910
|
+
// reads its own hub-bounded set (see traverse.ts) — so the evidence
|
|
911
|
+
// pick may lie outside the cap, and it is still yielded FIRST on
|
|
912
|
+
// purpose. The cap bounds what the hop EXPLORES; it must never make
|
|
913
|
+
// the evidence-ranked continuation unreachable, or the derivation the
|
|
914
|
+
// corpus corroborates would lose to exploration order. Nothing is
|
|
915
|
+
// reallocated: `nx` is walked with a skip-in-place for the duplicate.
|
|
858
916
|
const preferred = nx.length > 1
|
|
859
917
|
? this.host.chooseNext?.(it.node)
|
|
860
918
|
: undefined;
|
|
@@ -1038,12 +1096,12 @@ export class GraphSearch {
|
|
|
1038
1096
|
const memo = this.recompleteMemo;
|
|
1039
1097
|
if (memo.has(node)) return memo.get(node) ?? null;
|
|
1040
1098
|
// Re-covering is how a PRODUCED node's bytes enter the search at all: the
|
|
1041
|
-
// cover machinery otherwise only ever sees the QUERY's spans.
|
|
1099
|
+
// cover machinery otherwise only ever sees the QUERY's spans. The recursion
|
|
1042
1100
|
// is allowed to nest — a chain IS nested completions — but it is bounded so
|
|
1043
|
-
// the work stays the ANSWER's (
|
|
1044
|
-
// guard, only ACCEPTED completions recurse, and the nested solve
|
|
1045
|
-
// the form by its own shape instead of re-recognising the
|
|
1046
|
-
// inside it.
|
|
1101
|
+
// the work stays the ANSWER's (bounded-reads.md): the stack below is the
|
|
1102
|
+
// cycle guard, only ACCEPTED completions recurse, and the nested solve
|
|
1103
|
+
// decomposes the form by its own shape instead of re-recognising the
|
|
1104
|
+
// corpus's hub forms inside it.
|
|
1047
1105
|
//
|
|
1048
1106
|
// `recompleteOpen` IS the stack of the chain being built, so MEMBERSHIP is
|
|
1049
1107
|
// the cycle guard: a node already open on this chain cannot re-enter it.
|
|
@@ -1079,10 +1137,48 @@ export class GraphSearch {
|
|
|
1079
1137
|
// a deeper rewrite chain is made of.
|
|
1080
1138
|
const rec = this.host.recogniseSpan(bytes);
|
|
1081
1139
|
const kids = new Set(nrec.kids);
|
|
1140
|
+
// THE NODE'S OWN KIDS ARE SITES BY STRUCTURE — recognition cannot be the
|
|
1141
|
+
// only source. A produced composite is decomposed by ITS OWN SHAPE, and
|
|
1142
|
+
// at hub scale the recognition of a produced span returns the WHOLE while
|
|
1143
|
+
// deliberately suppressing its atoms (the off-boundary suppression), so
|
|
1144
|
+
// the kid filter below would admit nothing at all and the chain would end
|
|
1145
|
+
// at the intermediate composite. Measured on the chain
|
|
1146
|
+
// `seed → "p q" → (p→r, q→s) → "r s" → "m n"`: below the flip it reaches
|
|
1147
|
+
// "m n" with fuse+recompose, above it stops at "p q" — the trace shows
|
|
1148
|
+
// `recognise("p q") ⇒ form "p q"` alone, no parts.
|
|
1149
|
+
//
|
|
1150
|
+
// Laying the node's kids out over its own bytes restores exactly the
|
|
1151
|
+
// decomposition the node's tree already states; a kid recognition ALREADY
|
|
1152
|
+
// offers is skipped, so below the flip the seed set is byte-identical to
|
|
1153
|
+
// what it was. The filter's guarantee is untouched: nothing beyond the
|
|
1154
|
+
// node's own kids may enter.
|
|
1155
|
+
const recognised = rec.sites.filter((s) => kids.has(s.payload));
|
|
1156
|
+
// DEDUPED BY SPAN, not by payload. A node whose kids repeat (`"abab"`
|
|
1157
|
+
// folds as ["ab","ab"]) has TWO occurrences of the same node at different
|
|
1158
|
+
// offsets; a payload-keyed set suppressed the structural site for BOTH, so
|
|
1159
|
+
// the second occurrence had no site at all. The set now holds the spans
|
|
1160
|
+
// recognition already covers, and a structural site is added exactly when
|
|
1161
|
+
// nothing covers that PLACE — O(1) lookups, no extra reads.
|
|
1162
|
+
const seenSpans = new Set(recognised.map((s) => `${s.start}:${s.end}`));
|
|
1163
|
+
const structural: Site[] = [];
|
|
1164
|
+
{
|
|
1165
|
+
let off = 0;
|
|
1166
|
+
for (const k of nrec.kids) {
|
|
1167
|
+
const len = this.store.bytesPrefix(k, ALL).length;
|
|
1168
|
+
if (!seenSpans.has(`${off}:${off + len}`) && len > 0) {
|
|
1169
|
+
structural.push({
|
|
1170
|
+
start: off,
|
|
1171
|
+
end: Math.min(off + len, bytes.length),
|
|
1172
|
+
payload: k,
|
|
1173
|
+
});
|
|
1174
|
+
}
|
|
1175
|
+
off += len;
|
|
1176
|
+
}
|
|
1177
|
+
}
|
|
1082
1178
|
const solved = this.solve(
|
|
1083
1179
|
bytes.length,
|
|
1084
1180
|
{
|
|
1085
|
-
sites:
|
|
1181
|
+
sites: [...recognised, ...structural],
|
|
1086
1182
|
leaves: rec.leaves,
|
|
1087
1183
|
splits: rec.splits,
|
|
1088
1184
|
starts: rec.starts,
|
|
@@ -1162,6 +1258,14 @@ export class GraphSearch {
|
|
|
1162
1258
|
if (!this.host.recogniseSpan) return;
|
|
1163
1259
|
const tail = queryBytes.subarray(fact.j, queryLen);
|
|
1164
1260
|
if (tail.length === 0) return;
|
|
1261
|
+
// Report ONLY the invocations that could have joined: the search asks this
|
|
1262
|
+
// rule for every finalized out with a node, which includes the one-byte
|
|
1263
|
+
// outs the cover bridges with — measured, 68 refusals for a single
|
|
1264
|
+
// 3-relation query, all of them letters. A form shorter than one window is
|
|
1265
|
+
// not a fact a join could travel through, so it is not a refusal worth
|
|
1266
|
+
// reporting; W is the same line the rest of the mind draws between a chance
|
|
1267
|
+
// overlap and a form.
|
|
1268
|
+
const reportable = fact.bytes.length >= this.maxGroup;
|
|
1165
1269
|
// The entity candidates are the forms the fact's own bytes CONTAIN — the
|
|
1166
1270
|
// same recogniser the query went through, so the evidence standard is the
|
|
1167
1271
|
// query's. A byte atom is never a subject; the fact's own node is the span
|
|
@@ -1170,12 +1274,66 @@ export class GraphSearch {
|
|
|
1170
1274
|
// LENDS it when it can (Mind does, with the response-scoped struct cache),
|
|
1171
1275
|
// and a bare host falls back to the raw-store probe, so the search stays
|
|
1172
1276
|
// host-based.
|
|
1173
|
-
const
|
|
1174
|
-
|
|
1175
|
-
|
|
1176
|
-
? this.host.leadsSomewhere(
|
|
1177
|
-
: this.store.hasNext(
|
|
1277
|
+
const factRec = this.host.recogniseSpan(fact.bytes);
|
|
1278
|
+
const leads = (id: number): boolean =>
|
|
1279
|
+
this.host.leadsSomewhere !== undefined
|
|
1280
|
+
? this.host.leadsSomewhere(id)
|
|
1281
|
+
: this.store.hasNext(id) || this.store.hasHalo(id);
|
|
1282
|
+
// THE QUERY'S OWN SUBJECT, CANONICALLY. The filter used raw bytes and the
|
|
1283
|
+
// store's nodes are canonical, so `Eiffel Tower country` in the query did
|
|
1284
|
+
// not match the deposited `eiffel tower country` — measured, that is the
|
|
1285
|
+
// trap's wrong answer.
|
|
1286
|
+
//
|
|
1287
|
+
// TAKEN FROM THE RECOGNITION THE RESPONSE ALREADY COMPUTED — the host's
|
|
1288
|
+
// `recogniseSpan`, the same surface the rest of this search uses — not from
|
|
1289
|
+
// a second offset scan. `canonicalQueryNodes` re-derived, per byte offset,
|
|
1290
|
+
// what `recognise` had already resolved once per query (its memo is keyed by
|
|
1291
|
+
// content), and that scan was the largest single cost the DIANOT join added:
|
|
1292
|
+
// measured against the pre-change tree, the same fixture and the same test
|
|
1293
|
+
// were 21 s slower with the scan than without it. A recognised site IS a
|
|
1294
|
+
// canonical node of the query that can lead somewhere, which is exactly the
|
|
1295
|
+
// set this filter wants, and it costs nothing to read.
|
|
1296
|
+
const queryNodes = new Set<number>(
|
|
1297
|
+
(this.host.recogniseSpan?.(queryBytes)?.sites ?? []).map((s) =>
|
|
1298
|
+
s.payload
|
|
1299
|
+
),
|
|
1178
1300
|
);
|
|
1301
|
+
// TWO SOURCES, ONE ADMISSION. The recognition of a STORED WHOLE returns the
|
|
1302
|
+
// whole and stops — measured: for `The director of Eva is Gustaf Molander.`
|
|
1303
|
+
// it yields exactly ONE site, the fact's own node — so the entity a join
|
|
1304
|
+
// exists for is never proposed. The canonical fold is the second source,
|
|
1305
|
+
// and the scan runs only for a FORM (≥ W: a one-byte out is not something to
|
|
1306
|
+
// join through, and running it per letter measured 20-26 s in test/99).
|
|
1307
|
+
const W = this.maxGroup;
|
|
1308
|
+
const proposed = new Map<number, Uint8Array>();
|
|
1309
|
+
// The SOURCE of each proposal travels with it: a refusal that names only the
|
|
1310
|
+
// bytes leaves the next reader guessing which path proposed them — three
|
|
1311
|
+
// attempts at the chained join were spent fixing paths that never produced
|
|
1312
|
+
// the offending candidate.
|
|
1313
|
+
const source = new Map<number, string>();
|
|
1314
|
+
for (const s of factRec.sites) {
|
|
1315
|
+
if (s.payload >= 0 && leads(s.payload)) {
|
|
1316
|
+
proposed.set(s.payload, this.store.bytesPrefix(s.payload, ALL));
|
|
1317
|
+
source.set(s.payload, "recognised site");
|
|
1318
|
+
}
|
|
1319
|
+
}
|
|
1320
|
+
if (this.host.canonResolve !== undefined && fact.bytes.length >= W) {
|
|
1321
|
+
const canon = this.host.canonResolve.bind(this.host);
|
|
1322
|
+
for (let start = 0; start < fact.bytes.length; start++) {
|
|
1323
|
+
for (let end = fact.bytes.length; end - start >= W; end--) {
|
|
1324
|
+
const id = canon(fact.bytes.subarray(start, end));
|
|
1325
|
+
if (id === null) continue;
|
|
1326
|
+
if (leads(id)) {
|
|
1327
|
+
proposed.set(id, this.store.bytesPrefix(id, ALL));
|
|
1328
|
+
source.set(id, "canonical fold");
|
|
1329
|
+
}
|
|
1330
|
+
break; // the longest form at this offset wins
|
|
1331
|
+
}
|
|
1332
|
+
}
|
|
1333
|
+
}
|
|
1334
|
+
const leading = [...proposed]
|
|
1335
|
+
.filter(([payload]) => payload !== fact.node && !queryNodes.has(payload))
|
|
1336
|
+
.map(([payload, bytes]) => ({ payload, bytes }));
|
|
1179
1337
|
// …then prefer the entity the query did NOT name, and the MAXIMAL one. The
|
|
1180
1338
|
// join exists to reach the subject the query never wrote, so:
|
|
1181
1339
|
// • a candidate the query already contains is the query's OWN subject, and
|
|
@@ -1186,32 +1344,104 @@ export class GraphSearch {
|
|
|
1186
1344
|
// introduces ("Timur" must not win over "Timur Bekmambetov").
|
|
1187
1345
|
// Byte work over bytes already read, and the pruning REMOVES the
|
|
1188
1346
|
// resolve()/nextFirst() probes these candidates would have paid.
|
|
1347
|
+
if (leading.length === 0) {
|
|
1348
|
+
if (this.host.meter) this.host.meter.joinNoEntity++;
|
|
1349
|
+
if (reportable) {
|
|
1350
|
+
// Report WHAT the recognition returned, not just that nothing led: the
|
|
1351
|
+
// count and the first few site texts are the difference between "the
|
|
1352
|
+
// fact was not recognised" and "it was recognised but nothing led".
|
|
1353
|
+
const seen = factRec.sites.slice(0, 3).map((s) =>
|
|
1354
|
+
this.store.bytesPrefix(s.payload, ALL)
|
|
1355
|
+
);
|
|
1356
|
+
this.host.reportSearch?.(
|
|
1357
|
+
"deriveThroughMiss",
|
|
1358
|
+
[fact.bytes, tail, ...seen],
|
|
1359
|
+
`no entity inside the fact leads anywhere — ${factRec.sites.length} site(s) recognised inside it`,
|
|
1360
|
+
);
|
|
1361
|
+
}
|
|
1362
|
+
}
|
|
1189
1363
|
const candidates = leading
|
|
1190
|
-
.map((s) => ({
|
|
1191
|
-
payload: s.payload,
|
|
1192
|
-
bytes: this.store.bytesPrefix(s.payload, ALL),
|
|
1193
|
-
}))
|
|
1194
|
-
.filter((c) => indexOf(queryBytes, c.bytes, 0) < 0)
|
|
1195
1364
|
.filter((c, _i, all) =>
|
|
1196
1365
|
!all.some((o) =>
|
|
1197
1366
|
o.bytes.length > c.bytes.length && indexOf(o.bytes, c.bytes, 0) >= 0
|
|
1198
1367
|
)
|
|
1199
1368
|
);
|
|
1200
1369
|
for (const c of candidates) {
|
|
1201
|
-
|
|
1202
|
-
|
|
1203
|
-
|
|
1204
|
-
|
|
1370
|
+
// THE SHORTEST TAIL PREFIX WHOSE KEY ALSO LEADS SOMEWHERE.
|
|
1371
|
+
//
|
|
1372
|
+
// The conclusion covers only that prefix, so the rest of the tail stays
|
|
1373
|
+
// for the step after — which is what CHAINING is (a whole-tail key
|
|
1374
|
+
// consumed the whole remainder and made every join terminal: measured,
|
|
1375
|
+
// `joinFired=0` on a three-relation query).
|
|
1376
|
+
//
|
|
1377
|
+
// A KEY THAT RESOLVES IS NOT ENOUGH. Measured with a dry run of these
|
|
1378
|
+
// very primitives: for the candidate `Sweden` the first tail prefix that
|
|
1379
|
+
// resolves is `" "` — the key `Sweden ` (a trailing space) — and it leads
|
|
1380
|
+
// NOWHERE (`nextFirst` = 0), so accepting it refused the join while the
|
|
1381
|
+
// key that names the fact (`sweden capital`) sat one prefix further. The
|
|
1382
|
+
// loop therefore asks BOTH questions before accepting, and keeps looking
|
|
1383
|
+
// otherwise.
|
|
1384
|
+
//
|
|
1385
|
+
// EXACT FIRST, THEN CANONICAL: the corpus holds both identities.
|
|
1386
|
+
let key: number | null = null;
|
|
1387
|
+
let used = 0;
|
|
1388
|
+
// The continuation the accepted key leads to, carried out of the loop:
|
|
1389
|
+
// the loop already HAD to read it to accept the key (a key that leads
|
|
1390
|
+
// nowhere is not the relation), so re-reading it after the loop was a
|
|
1391
|
+
// duplicate read and a branch that could never be taken.
|
|
1392
|
+
let next: number | null = null;
|
|
1393
|
+
let keyBytes = c.bytes;
|
|
1394
|
+
// THE PREFIX ENDS ARE THE TAIL'S OWN FOLD BOUNDARIES, not every byte
|
|
1395
|
+
// length. The key is `entity + prefix`, and the prefix that names a
|
|
1396
|
+
// stored relation ends where the fold cuts: measured over four join-firing
|
|
1397
|
+
// queries, 5 of 5 accepted keys ended on a boundary (or the tail's end)
|
|
1398
|
+
// while the byte-by-byte scan spent 153 probes where 14 boundaries would
|
|
1399
|
+
// do. Same criterion — resolves AND leads — same shortest-first order, so
|
|
1400
|
+
// the answer is the same one the enumeration found; only the candidates
|
|
1401
|
+
// come from the structure instead of from the byte count. A host with no
|
|
1402
|
+
// boundary rule falls back to the enumeration.
|
|
1403
|
+
const cuts = this.host.contentCuts?.(tail);
|
|
1404
|
+
const ends = cuts && cuts.length > 0
|
|
1405
|
+
? [...cuts.filter((c) => c > 0 && c < tail.length), tail.length]
|
|
1406
|
+
: Array.from({ length: tail.length }, (_, i) => i + 1);
|
|
1407
|
+
for (const len of ends) {
|
|
1408
|
+
keyBytes = concat2(c.bytes, tail.subarray(0, len));
|
|
1409
|
+
const k = this.host.resolve(keyBytes) ??
|
|
1410
|
+
this.host.canonResolve?.(keyBytes) ??
|
|
1411
|
+
null;
|
|
1412
|
+
if (k === null) continue;
|
|
1413
|
+
const nx = this.store.nextFirst(k, 1);
|
|
1414
|
+
if (nx.length === 0) continue;
|
|
1415
|
+
key = k;
|
|
1416
|
+
used = len;
|
|
1417
|
+
next = nx[0];
|
|
1418
|
+
break;
|
|
1419
|
+
}
|
|
1420
|
+
if (key === null) {
|
|
1421
|
+
if (this.host.meter) this.host.meter.joinNoKey++;
|
|
1422
|
+
if (reportable) {
|
|
1423
|
+
this.host.reportSearch?.(
|
|
1424
|
+
"deriveThroughMiss",
|
|
1425
|
+
[c.bytes, tail, keyBytes],
|
|
1426
|
+
`no learnt key names this entity and tail together ` +
|
|
1427
|
+
`(candidate #${c.payload}, from the ${
|
|
1428
|
+
source.get(c.payload) ?? "unknown"
|
|
1429
|
+
} source)`,
|
|
1430
|
+
);
|
|
1431
|
+
}
|
|
1432
|
+
continue;
|
|
1433
|
+
}
|
|
1434
|
+
if (this.host.meter) this.host.meter.joinFired++;
|
|
1205
1435
|
yield {
|
|
1206
1436
|
premises: [fact],
|
|
1207
1437
|
conclusion: {
|
|
1208
1438
|
kind: "out",
|
|
1209
1439
|
i: fact.i,
|
|
1210
|
-
j:
|
|
1211
|
-
bytes: this.store.bytesPrefix(
|
|
1440
|
+
j: fact.j + used,
|
|
1441
|
+
bytes: this.store.bytesPrefix(next!, ALL),
|
|
1212
1442
|
cover: true,
|
|
1213
1443
|
rec: true,
|
|
1214
|
-
node:
|
|
1444
|
+
node: next!,
|
|
1215
1445
|
throughFact: true,
|
|
1216
1446
|
},
|
|
1217
1447
|
cost: STEP,
|
package/src/mind/index.ts
CHANGED
|
@@ -4,7 +4,12 @@
|
|
|
4
4
|
// exported from mind/mind.ts directly.
|
|
5
5
|
|
|
6
6
|
export { Mind } from "./mind.js";
|
|
7
|
-
export type {
|
|
7
|
+
export type {
|
|
8
|
+
CorpusTextPair,
|
|
9
|
+
CorpusTextResult,
|
|
10
|
+
Input,
|
|
11
|
+
Response,
|
|
12
|
+
} from "./mind.js";
|
|
8
13
|
export type { ComputedSpan, ExtensionHost } from "./mind.js";
|
|
9
14
|
export type {
|
|
10
15
|
MechanismResult,
|
|
@@ -38,3 +43,5 @@ export type {
|
|
|
38
43
|
NarrowDecisionData,
|
|
39
44
|
Provenance,
|
|
40
45
|
} from "./pipeline.js";
|
|
46
|
+
export { sampleCorpus, searchCorpus } from "./corpus.js";
|
|
47
|
+
export type { CorpusMiss, CorpusPair, CorpusResult } from "./corpus.js";
|
package/src/mind/junction.ts
CHANGED
|
@@ -208,7 +208,7 @@ function cachedContainers(
|
|
|
208
208
|
* is legitimately reached across many containing structures. Half the
|
|
209
209
|
* successful junctions would be lost.
|
|
210
210
|
*
|
|
211
|
-
*
|
|
211
|
+
* REFUTED EARLY-STOP (side-cone exhaustion, saturation.md's "real saturation"):
|
|
212
212
|
* stopping the walk the moment ONE side's upward cone is emptied is wrong,
|
|
213
213
|
* in both a hub-guarded form and a hub-flagged form. The junction test is
|
|
214
214
|
* a BYTE containment over the UNION of the two cones, and a junction can be
|
|
@@ -265,13 +265,13 @@ export function junctionContainersFrom(
|
|
|
265
265
|
d: 0,
|
|
266
266
|
}));
|
|
267
267
|
while (stack.length > 0 && out.length < bound) {
|
|
268
|
-
// BUDGET EXHAUSTION IS AN ABSTENTION, AND IT MUST BE VISIBLE
|
|
269
|
-
// walk stops with work still on the stack, the caller
|
|
270
|
-
// and falls through to a lower ladder rung —
|
|
271
|
-
// outside, from a walk that looked everywhere
|
|
272
|
-
// SHARED budget (cross-region's one k·W allowance
|
|
273
|
-
// pair can drain it, so a later pair's exact tier may
|
|
274
|
-
// this counter is the only thing that says so.
|
|
268
|
+
// BUDGET EXHAUSTION IS AN ABSTENTION, AND IT MUST BE VISIBLE
|
|
269
|
+
// (INVARIANTS.md). The walk stops with work still on the stack, the caller
|
|
270
|
+
// reads "no container" and falls through to a lower ladder rung —
|
|
271
|
+
// indistinguishable, from the outside, from a walk that looked everywhere
|
|
272
|
+
// and found nothing. With a SHARED budget (cross-region's one k·W allowance
|
|
273
|
+
// per tier) an EARLIER pair can drain it, so a later pair's exact tier may
|
|
274
|
+
// never run at all; this counter is the only thing that says so.
|
|
275
275
|
if (b.n-- <= 0) {
|
|
276
276
|
if (ctx.meter) ctx.meter.junctionBudgetExhausted++;
|
|
277
277
|
break;
|
package/src/mind/learning.ts
CHANGED
|
@@ -306,7 +306,8 @@ function constituentSketch(ctx: MindContext, id: number, k: number): number[] {
|
|
|
306
306
|
for (const g of constituentSketch(ctx, kid, k)) pool.push(g);
|
|
307
307
|
}
|
|
308
308
|
}
|
|
309
|
-
// Bottom-k by identity, then by id so ties are corpus-determined
|
|
309
|
+
// Bottom-k by identity, then by id so ties are corpus-determined
|
|
310
|
+
// (determinism.md).
|
|
310
311
|
pool.sort((a, b) => (unitPriority(a) - unitPriority(b)) || (a - b));
|
|
311
312
|
const seen = new Set<number>();
|
|
312
313
|
out = [];
|
|
@@ -368,15 +369,15 @@ function constituentSketch(ctx: MindContext, id: number, k: number): number[] {
|
|
|
368
369
|
* terms unique to that partner, which dilute but never mislead; the shared
|
|
369
370
|
* units contribute the signal.
|
|
370
371
|
*
|
|
371
|
-
* HUBS ARE THE ONE EXCLUSION, read LIMITed as `parentsFirst(n, bound+1)` —
|
|
372
|
-
*
|
|
373
|
-
*
|
|
374
|
-
*
|
|
375
|
-
*
|
|
376
|
-
*
|
|
377
|
-
*
|
|
378
|
-
*
|
|
379
|
-
*
|
|
372
|
+
* HUBS ARE THE ONE EXCLUSION, read LIMITed as `parentsFirst(n, bound+1)` — the
|
|
373
|
+
* store's own exact hub-or-not probe (a result longer than the bound means MORE
|
|
374
|
+
* than the bound), never a fan-in-sized read. A constituent with more than √N
|
|
375
|
+
* structural parents is scaffolding by bounded-reads.md's bound: " is ", "the
|
|
376
|
+
* ". Superposing it would put a term shared by every deposit into every
|
|
377
|
+
* profile, ALL halos would correlate, and the concept threshold's null model
|
|
378
|
+
* (unrelated halos at 0 ± 1/√D) that halo-sketch.md's hygiene note protects
|
|
379
|
+
* would collapse. It is still DESCENDED into — a hub chunk can contain a rare
|
|
380
|
+
* unit — but contributes nothing itself.
|
|
380
381
|
*
|
|
381
382
|
* Byte atoms are skipped in BOTH representations (a negative id and a stored
|
|
382
383
|
* kid-less node): an atom's fan-in is the alphabet's, so it can only ever
|
|
@@ -386,32 +387,32 @@ function constituentSketch(ctx: MindContext, id: number, k: number): number[] {
|
|
|
386
387
|
* analogy strength 0.3636 -> 0.2004, "no halo-tier company evidence",
|
|
387
388
|
* test/29 C1).
|
|
388
389
|
*
|
|
389
|
-
* A FUNCTION OF THE NODE AND THE CORPUS STATE — stated precisely, because
|
|
390
|
-
*
|
|
391
|
-
*
|
|
392
|
-
*
|
|
393
|
-
*
|
|
394
|
-
*
|
|
395
|
-
*
|
|
396
|
-
*
|
|
397
|
-
*
|
|
398
|
-
*
|
|
399
|
-
*
|
|
400
|
-
*
|
|
401
|
-
*
|
|
402
|
-
*
|
|
403
|
-
*
|
|
404
|
-
*
|
|
390
|
+
* A FUNCTION OF THE NODE AND THE CORPUS STATE — stated precisely, because the
|
|
391
|
+
* weaker claim is the true one. The constituents are read from the STORE, never
|
|
392
|
+
* from the depositing tree's id map: that map holds only the nodes THIS deposit
|
|
393
|
+
* newly interned, so a partner met a second time yielded a profile missing
|
|
394
|
+
* exactly those constituents, the exact-partner case fell from cosine 1 to
|
|
395
|
+
* 1/√(1+k), and the geometry stopped meaning anything. Reading the store fixes
|
|
396
|
+
* that. It does NOT make the profile permanent: the hub test reads fan-in
|
|
397
|
+
* against √N and both grow with training, so a partner poured early and again
|
|
398
|
+
* late can profile differently. That residue is confined to the hub EXCLUSION —
|
|
399
|
+
* which terms are dropped as scaffolding — and never to which units are found,
|
|
400
|
+
* because the descent itself is now order-independent. The drift is
|
|
401
|
+
* one-directional and benign: a term can only ever go from contributing to
|
|
402
|
+
* being excluded as scaffolding. Replay of a fixed training order is
|
|
403
|
+
* bit-identical, so determinism.md holds. What must not be claimed is that a
|
|
404
|
+
* node's profile is fixed for all time; it is fixed given the corpus that has
|
|
405
|
+
* been seen.
|
|
405
406
|
*
|
|
406
|
-
*
|
|
407
|
-
*
|
|
408
|
-
*
|
|
409
|
-
*
|
|
410
|
-
*
|
|
411
|
-
*
|
|
412
|
-
*
|
|
413
|
-
*
|
|
414
|
-
*
|
|
407
|
+
* THE NULL MODEL IS OTHERWISE UNTOUCHED (halo-sketch.md). Every term is still a
|
|
408
|
+
* seeded function of a NODE IDENTITY, never a gist, so no byte-similarity
|
|
409
|
+
* between partners can leak content similarity into distributional similarity.
|
|
410
|
+
* The result is normalized, so ONE episode still pours ONE unit of mass: {@link
|
|
411
|
+
* Store.haloMass} keeps counting episodes and every mass-based reading is
|
|
412
|
+
* unchanged. Two partners sharing j of k discriminating constituents meet at
|
|
413
|
+
* j/(1+k) — graded evidence, above the 1/√D noise floor and below
|
|
414
|
+
* conceptThreshold until the overlap is most of the content, which is the
|
|
415
|
+
* semantics "same company" should have.
|
|
415
416
|
*
|
|
416
417
|
* Bounded: at most {@link PROFILE_VISITS} constituents are classified, each
|
|
417
418
|
* by ONE LIMITed structural-parent read, so a pour costs O(1) reads in the
|