@hviana/sema 0.8.1 → 0.8.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +29 -29
- package/TRADEMARKS.md +0 -1
- package/dist/src/config.d.ts +28 -0
- package/dist/src/config.js +20 -0
- package/dist/src/geometry.d.ts +21 -0
- package/dist/src/geometry.js +21 -0
- package/dist/src/meter.d.ts +76 -0
- package/dist/src/meter.js +95 -0
- package/dist/src/mind/attention.d.ts +4 -0
- package/dist/src/mind/attention.js +165 -16
- package/dist/src/mind/canonical.d.ts +16 -0
- package/dist/src/mind/canonical.js +41 -0
- package/dist/src/mind/corpus.d.ts +40 -0
- package/dist/src/mind/corpus.js +149 -0
- package/dist/src/mind/graph-search.d.ts +7 -0
- package/dist/src/mind/graph-search.js +254 -24
- package/dist/src/mind/index.d.ts +3 -1
- package/dist/src/mind/index.js +1 -0
- package/dist/src/mind/match.d.ts +9 -4
- package/dist/src/mind/match.js +147 -61
- package/dist/src/mind/mechanisms/cast.js +19 -3
- package/dist/src/mind/mechanisms/confluence.js +24 -0
- package/dist/src/mind/mechanisms/cover.js +6 -0
- package/dist/src/mind/mechanisms/recall.js +32 -4
- package/dist/src/mind/mind.d.ts +57 -0
- package/dist/src/mind/mind.js +72 -1
- package/dist/src/mind/pipeline-mechanism.d.ts +7 -0
- package/dist/src/mind/pipeline.js +66 -20
- package/dist/src/mind/primitives.js +9 -1
- package/dist/src/mind/rationale.d.ts +28 -1
- package/dist/src/mind/rationale.js +22 -1
- package/dist/src/mind/reasoning.d.ts +25 -3
- package/dist/src/mind/reasoning.js +125 -20
- package/dist/src/mind/recognition.js +4 -8
- package/dist/src/mind/resonance.js +20 -1
- package/dist/src/mind/trace.js +1 -0
- package/dist/src/mind/traverse.js +15 -3
- package/dist/src/mind/types.d.ts +49 -4
- package/docs/INVARIANTS.md +2 -2
- package/docs/architecture/bounded-reads.md +1 -1
- package/docs/architecture/commonality.md +2 -2
- package/docs/architecture/cost-model.md +2 -2
- package/docs/architecture/determinism.md +7 -7
- package/docs/architecture/match-project.md +2 -3
- package/docs/architecture/mechanism-market.md +10 -10
- package/docs/architecture/meter.md +5 -5
- package/docs/architecture/store.md +3 -3
- package/docs/failures/tempting-but-wrong.md +34 -6
- package/docs/harness/gates.md +2 -2
- package/docs/mechanisms/cast.md +2 -2
- package/docs/mechanisms/cover.md +2 -3
- package/docs/mechanisms/extraction.md +7 -7
- package/docs/mechanisms/recall.md +8 -9
- package/jsr.json +1 -1
- package/package.json +1 -1
- package/src/alu/README.md +11 -12
- package/src/config.ts +48 -0
- package/src/geometry.ts +21 -0
- package/src/meter.ts +98 -0
- package/src/mind/attention.ts +167 -16
- package/src/mind/canonical.ts +43 -0
- package/src/mind/corpus.ts +202 -0
- package/src/mind/graph-search.ts +277 -23
- package/src/mind/index.ts +8 -1
- package/src/mind/match.ts +148 -57
- package/src/mind/mechanisms/cast.ts +20 -2
- package/src/mind/mechanisms/confluence.ts +24 -0
- package/src/mind/mechanisms/cover.ts +5 -0
- package/src/mind/mechanisms/recall.ts +32 -4
- package/src/mind/mind.ts +125 -0
- package/src/mind/pipeline-mechanism.ts +7 -0
- package/src/mind/pipeline.ts +79 -22
- package/src/mind/primitives.ts +9 -1
- package/src/mind/rationale.ts +35 -1
- package/src/mind/reasoning.ts +145 -13
- package/src/mind/recognition.ts +4 -8
- package/src/mind/resonance.ts +19 -1
- package/src/mind/trace.ts +1 -0
- package/src/mind/traverse.ts +16 -6
- package/src/mind/types.ts +53 -4
- package/test/100-complete-grounding-trace.test.mjs +109 -0
- package/test/101-alignment-gap-bound.test.mjs +106 -0
- package/test/102-production-composes-at-scale.test.mjs +110 -0
- package/test/103-alignment-gap-budget.test.mjs +89 -0
- package/test/104-composition-is-reported.test.mjs +90 -0
- package/test/105-derive-through-reports-its-refusal.test.mjs +137 -0
- package/test/106-the-join-fires.test.mjs +94 -0
- package/test/107-the-join-is-counted.test.mjs +81 -0
- package/test/108-the-join-chains.test.mjs +78 -0
- package/test/109-the-pivot-is-counted.test.mjs +60 -0
- package/test/110-the-reasoner-stops-when-the-question-is-answered.test.mjs +91 -0
- package/test/111-the-cover-assembly-is-counted.test.mjs +74 -0
- package/test/112-the-exploration-does-not-grow-with-the-hub.test.mjs +89 -0
- package/test/113-the-rationale-payload-is-bounded.test.mjs +84 -0
- package/test/114-alignment-budget-is-per-sweep.test.mjs +93 -0
- package/test/116-the-extension-is-gated-by-the-pipelines-own-remainder.test.mjs +100 -0
- package/test/117-corpus-search.test.mjs +171 -0
- package/test/118-the-join-reaches-a-key-off-the-cut.test.mjs +74 -0
- package/test/119-the-work-does-not-grow-with-the-corpus.test.mjs +122 -0
- package/test/120-composition-is-consequence.test.mjs +132 -0
- package/test/121-the-extension-does-not-grow-with-the-corpus.test.mjs +128 -0
- package/test/122-the-climb-search-does-not-grow-with-the-corpus.test.mjs +117 -0
- package/test/123-the-paired-formulas-agree.test.mjs +90 -0
- package/test/125-the-post-grounding-branch-publishes-its-operand.test.mjs +51 -0
- package/test/126-the-pipeline-does-not-name-mechanisms.test.mjs +42 -0
- package/test/128-the-leads-somewhere-pair-agrees.test.mjs +83 -0
- package/test/129-the-trace-payload-shape.test.mjs +164 -0
- package/test/14-scaling.test.mjs +10 -7
- package/test/32-confluence.test.mjs +68 -0
- package/test/38-reason-restate-guard.test.mjs +8 -2
- package/test/43-cast-analog-seat.test.mjs +10 -0
- package/test/55-cost-meter.test.mjs +859 -0
- package/test/76-reference-binding.test.mjs +6 -1
- package/test/89-completion-recursion.test.mjs +30 -5
|
@@ -775,15 +775,18 @@ export async function voteRegions(ctx, query, regions, k, mode, N, reachMemo, td
|
|
|
775
775
|
}
|
|
776
776
|
contrastiveMargin = margin;
|
|
777
777
|
// Scaled by what this region does NOT address — see `cov` above.
|
|
778
|
-
|
|
779
|
-
|
|
778
|
+
// The bar THIS gate applies: the estimator's noise scaled by what the
|
|
779
|
+
// region does NOT address (`cov`). ONE definition, used by the rejection
|
|
780
|
+
// path below and by the voted payload — the trace reports the applied bar.
|
|
781
|
+
const appliedFloor = estimatorNoise(ctx.store.D) * (1 - cov);
|
|
782
|
+
if (margin <= appliedFloor) {
|
|
780
783
|
recordRegion("contrastive-margin-rejection", {
|
|
781
784
|
selected,
|
|
782
785
|
reachNode: voterId,
|
|
783
786
|
idf,
|
|
784
787
|
dfWeight: wf,
|
|
785
788
|
contrastiveMargin: margin,
|
|
786
|
-
contrastiveNoiseFloor:
|
|
789
|
+
contrastiveNoiseFloor: appliedFloor,
|
|
787
790
|
...(contrastiveRival ? { contrastiveRival } : {}),
|
|
788
791
|
});
|
|
789
792
|
continue;
|
|
@@ -834,7 +837,12 @@ export async function voteRegions(ctx, query, regions, k, mode, N, reachMemo, td
|
|
|
834
837
|
...(contrastiveMargin !== undefined
|
|
835
838
|
? {
|
|
836
839
|
contrastiveMargin,
|
|
837
|
-
|
|
840
|
+
// THE BAR THE GATE ACTUALLY APPLIED — the same expression
|
|
841
|
+
// the rejection path's `appliedFloor` defines, inline here because
|
|
842
|
+
// this payload is built in a scope that does not carry that local.
|
|
843
|
+
// Publishing the raw estimatorNoise(D) instead made a region that
|
|
844
|
+
// PASSED look closer to its limit than it was.
|
|
845
|
+
contrastiveNoiseFloor: estimatorNoise(ctx.store.D) * (1 - cov),
|
|
838
846
|
...(contrastiveRival ? { contrastiveRival } : {}),
|
|
839
847
|
}
|
|
840
848
|
: {}),
|
|
@@ -939,7 +947,17 @@ export function poolVotes(ctx, regionVotes, sat, N, td) {
|
|
|
939
947
|
},
|
|
940
948
|
pool,
|
|
941
949
|
};
|
|
942
|
-
|
|
950
|
+
// THE SEARCH WAS THE ONE LAYER WITH NO TIME. The climb's phases are timed
|
|
951
|
+
// (voteRegions, structuralResonance, crossRegion) but the pooled derivation
|
|
952
|
+
// was not, so any cost or gain inside it stayed invisible.
|
|
953
|
+
// `timeSync`, not `time`: the search is SYNCHRONOUS, and wrapping it in a
|
|
954
|
+
// promise only to time it would make the profiled path wait where the
|
|
955
|
+
// unprofiled one does not (meter.ts's own contract).
|
|
956
|
+
if (ctx.meter) {
|
|
957
|
+
ctx.meter.timeSync("climb.derivation", () => lightestDerivation(system));
|
|
958
|
+
}
|
|
959
|
+
else
|
|
960
|
+
lightestDerivation(system);
|
|
943
961
|
const votes = new Map();
|
|
944
962
|
const votesIdf = new Map();
|
|
945
963
|
const support = new Map();
|
|
@@ -973,12 +991,24 @@ export function poolVotes(ctx, regionVotes, sat, N, td) {
|
|
|
973
991
|
// The LARGEST single region's contribution to this anchor's pooled vote.
|
|
974
992
|
// The pool is a SUM (deliberately — see the pooling note above), so it says
|
|
975
993
|
// how much evidence there is in total, never whether any ONE place in the
|
|
976
|
-
// query carries evidence on its own.
|
|
977
|
-
//
|
|
978
|
-
//
|
|
979
|
-
//
|
|
980
|
-
//
|
|
981
|
-
//
|
|
994
|
+
// query carries evidence on its own. Recorded here, beside the count,
|
|
995
|
+
// because this is the only place the per-region contributions are still
|
|
996
|
+
// separable.
|
|
997
|
+
//
|
|
998
|
+
// THE BAR IS THE POOLED FLOOR, AND IT WAS ONCE CLAIMED OTHERWISE HERE.
|
|
999
|
+
// This comment used to say that holding an anchor to consensusFloor(N)
|
|
1000
|
+
// "prices ONE region's evidence", so comparing a six-region sum against it
|
|
1001
|
+
// was "a dimensional error". THAT WAS FALSE. `thresholds.md` §2 derives
|
|
1002
|
+
// `consensusFloor` as the POOLED-vote significance floor ("each region
|
|
1003
|
+
// contributes at most ln(N/c) <= ln(N); ln(N) + 1/2 demands ..."), and the
|
|
1004
|
+
// climb weights by IDF, so the sum and the floor are in ONE dimension —
|
|
1005
|
+
// which is exactly why `recall.ts` gates `forest[0].idfVote` against it and
|
|
1006
|
+
// why `commitVotes` does too. The other two weighting modes DO leave that
|
|
1007
|
+
// dimension (by at most ln 2, two-sided: `direct` deflates a region and
|
|
1008
|
+
// `combined` inflates it), and the gates therefore read the IDF sum, which
|
|
1009
|
+
// is mode-independent; `test/55` tests 19 and 20 pin both halves — the sum
|
|
1010
|
+
// as the reading the bar is derived for, and the absence of any gate
|
|
1011
|
+
// inversion across the three modes.
|
|
982
1012
|
const regionPeak = new Map();
|
|
983
1013
|
const steps = [];
|
|
984
1014
|
let order = 0;
|
|
@@ -1136,12 +1166,23 @@ export function commitVotes(ctx, pooled, sat, regions, regionVoter, N, td, cfg)
|
|
|
1136
1166
|
anchor,
|
|
1137
1167
|
vote,
|
|
1138
1168
|
peak: regionPeak.get(anchor) ?? 0,
|
|
1169
|
+
idfVote: votesIdf.get(anchor) ?? 0,
|
|
1139
1170
|
start: s.start,
|
|
1140
1171
|
end: s.end,
|
|
1141
1172
|
breadth: (regionSupport.get(anchor) ?? 0) / totalRegions,
|
|
1142
1173
|
clusters: countClusters(regionSpans.get(anchor) ?? [], ctx.space.maxGroup),
|
|
1143
1174
|
};
|
|
1144
1175
|
})
|
|
1176
|
+
// THE ORDER IS NOT A PREFERENCE: with equal evidence it decides ADMISSION,
|
|
1177
|
+
// through the stable sort and the first-come overlap absorption below.
|
|
1178
|
+
// Measured on test/34's corpus, query "red": the two candidates (`red
|
|
1179
|
+
// circle` and `red square`) carry IDENTICAL `vote` and IDENTICAL `idfVote`
|
|
1180
|
+
// (1.3863 each, three seeds), so this comparator leaves them tied and the
|
|
1181
|
+
// stable sort keeps the ENUMERATION order — which is corpus-determined and
|
|
1182
|
+
// admits `red square`, 60/60 seeds. Adding an id tie-break (`|| a.anchor -
|
|
1183
|
+
// b.anchor`) picks `red circle` instead and makes a single region reach the
|
|
1184
|
+
// JOINT context, which is the premise `test/34` exists to protect. The
|
|
1185
|
+
// gates read IDF; this line only decides who gets looked at first.
|
|
1145
1186
|
.sort((a, b) => b.vote - a.vote);
|
|
1146
1187
|
const overlaps = (a, b) => a.start < b.end && b.start < a.end;
|
|
1147
1188
|
// Read the root cut from the anchors the QUERY pointed at. A vote standing
|
|
@@ -1177,6 +1218,13 @@ export function commitVotes(ctx, pooled, sat, regions, regionVoter, N, td, cfg)
|
|
|
1177
1218
|
rank,
|
|
1178
1219
|
pooledVote: point.vote,
|
|
1179
1220
|
idfVote: votesIdf.get(point.anchor) ?? 0,
|
|
1221
|
+
// The LARGEST single-region contribution behind this anchor — the bar
|
|
1222
|
+
// recall's own gate reads (mechanisms/recall.ts: forest[0].peak > LN2),
|
|
1223
|
+
// and until now the only decision-making quantity the climb computed and
|
|
1224
|
+
// did not publish. `regionPeak` reached `ranked` (see its build below)
|
|
1225
|
+
// and stopped there. Published, not recomputed: the value is the one the
|
|
1226
|
+
// climb already carries.
|
|
1227
|
+
peak: point.peak,
|
|
1180
1228
|
candidateBreadth: regions.length,
|
|
1181
1229
|
contributingVotes: regionAxioms.get(point.anchor) ?? 0,
|
|
1182
1230
|
contributingEvidence: regionSupport.get(point.anchor) ?? 0,
|
|
@@ -1206,6 +1254,18 @@ export function commitVotes(ctx, pooled, sat, regions, regionVoter, N, td, cfg)
|
|
|
1206
1254
|
let passesConsensusFloor;
|
|
1207
1255
|
let pastLeadingSaturation;
|
|
1208
1256
|
let tiedWithDominant;
|
|
1257
|
+
// ── ONE OF THREE ADMISSIONS, AND THEY ARE NOT THE SAME READING ────────
|
|
1258
|
+
// This block admits by VOTES: per-region evidence pooled, gated on the
|
|
1259
|
+
// natural break and on consensusFloor, with the dominant allowed to bypass
|
|
1260
|
+
// both. `structuralResonance` admits by a MARGIN over the estimator's own
|
|
1261
|
+
// noise, and `crossRegionVotes` admits by STRUCTURE (which regions may pair
|
|
1262
|
+
// at all, with at least one side individually discriminative). Read
|
|
1263
|
+
// together they look like one policy written three times; they are three
|
|
1264
|
+
// different measurements of the same question ("is this evidence?"), and
|
|
1265
|
+
// unifying them would average three readings into one — the mistake
|
|
1266
|
+
// `extraction.ts` records as "do not unify the two into one machine".
|
|
1267
|
+
// What they DO share, and must keep sharing, is the discipline of deriving
|
|
1268
|
+
// every bar from D/W/N rather than choosing it (thresholds.md).
|
|
1209
1269
|
const rejectionReasons = [];
|
|
1210
1270
|
if (absorbed) {
|
|
1211
1271
|
status = "overlap";
|
|
@@ -1216,9 +1276,75 @@ export function commitVotes(ctx, pooled, sat, regions, regionVoter, N, td, cfg)
|
|
|
1216
1276
|
pastLeadingSaturation = pastLeading;
|
|
1217
1277
|
const vote = votesIdf.get(point.anchor) ?? 0;
|
|
1218
1278
|
if (roots.length === 0) {
|
|
1219
|
-
//
|
|
1220
|
-
//
|
|
1221
|
-
//
|
|
1279
|
+
// THE DOMINANCE PRIVILEGE, AND THE TENSION IT CARRIES (measured).
|
|
1280
|
+
//
|
|
1281
|
+
// The first non-overlapping candidate is DOMINANT: it bypasses both
|
|
1282
|
+
// vote gates below and grounds on its own; only the leading-saturation
|
|
1283
|
+
// gate still applies to it. The privilege is load-bearing — analogies,
|
|
1284
|
+
// substitutions and composed contexts are precisely candidates the
|
|
1285
|
+
// query does NOT contain, and the engine loses them without it.
|
|
1286
|
+
//
|
|
1287
|
+
// WHICH candidate receives it, though, is decided by this loop's ORDER.
|
|
1288
|
+
// That order comes from `ranked`, and when two candidates carry equal
|
|
1289
|
+
// evidence the stable sort preserves the ENUMERATION order, so the
|
|
1290
|
+
// privilege is allocated by an ordering rather than by a rule.
|
|
1291
|
+
//
|
|
1292
|
+
// Measured on test/34's corpus, query "red":
|
|
1293
|
+
//
|
|
1294
|
+
// 0:#77 vote=1.3863 idf=1.3863 [0,3) | 1:#49 vote=1.3863 idf=1.3863 [0,3)
|
|
1295
|
+
//
|
|
1296
|
+
// Both candidates (`red square` #77, `red circle` #49) have IDENTICAL
|
|
1297
|
+
// `vote` AND IDENTICAL `idfVote` over the SAME support span, so the
|
|
1298
|
+
// comparator leaves them tied, the second is absorbed as "overlap", and
|
|
1299
|
+
// the first grounds. With this build's enumeration order that first is
|
|
1300
|
+
// `red square`, 60/60 seeds, and `red circle` — the JOINT context — is
|
|
1301
|
+
// never reached by "red" alone. That is the premise test/34 exists to
|
|
1302
|
+
// protect: no single region reaches the joint context, which is what
|
|
1303
|
+
// makes the binding query unreachable without direct region
|
|
1304
|
+
// interaction.
|
|
1305
|
+
//
|
|
1306
|
+
// THE TENSION: the premise therefore holds BY ENUMERATION ORDER, not by
|
|
1307
|
+
// a rule, so any change to this ordering can move the privilege onto the
|
|
1308
|
+
// joint context and let one region reach it. Measured: adding
|
|
1309
|
+
// `|| a.anchor - b.anchor` — the lowest-id tie-break that AGENTS.md §2
|
|
1310
|
+
// sanctions as an equivalent corpus-determined tie-break — does exactly
|
|
1311
|
+
// that: "red" then attends to `red circle`, test/34 fails 6/1, and the
|
|
1312
|
+
// canonical suite reports 1 failure.
|
|
1313
|
+
//
|
|
1314
|
+
// TWO ATTEMPTS TO MAKE IT A RULE, BOTH REFUTED BY MEASUREMENT:
|
|
1315
|
+
//
|
|
1316
|
+
// 1. EVIDENCE SEPARATION. Grant the privilege only when the first
|
|
1317
|
+
// candidate's evidence is separated from the next distinct
|
|
1318
|
+
// candidate's by more than the co-dominant band (sqrt(k) *
|
|
1319
|
+
// estimatorNoise(D)). Refuted: that band exists to ADMIT the
|
|
1320
|
+
// anchors the estimator cannot separate from the dominant — its own
|
|
1321
|
+
// documented purpose — so withholding the privilege on ties removes
|
|
1322
|
+
// the very case it was written for. Suite: 4 failures (the two
|
|
1323
|
+
// co-dominant band laws, breadth/scale invariance, test/29 D2).
|
|
1324
|
+
//
|
|
1325
|
+
// 2. QUERY-OWNED CONTENT. Grant the privilege only to a candidate
|
|
1326
|
+
// that IS a recognised region's identity (regions.some(r => r.id ===
|
|
1327
|
+
// point.anchor)). Measured: for "circle" that identity IS the
|
|
1328
|
+
// ranked candidate, so the privilege stays and `circle` grounds; for
|
|
1329
|
+
// "red" the identity is the `red` node itself while the candidates
|
|
1330
|
+
// are the conjunctions, so neither is privileged; for "red then
|
|
1331
|
+
// circle" the composed context carries idf 3.958 and clears both
|
|
1332
|
+
// gates on its own evidence. All four control queries came out
|
|
1333
|
+
// right — and the suite: 10 failures, six of them in the
|
|
1334
|
+
// analogy/counterfactual/CAST suites ("an analogy still transfers
|
|
1335
|
+
// from a structure the query never names"; "a substitute the query
|
|
1336
|
+
// NAMES may still be voiced"). Refuted: the privilege exists to
|
|
1337
|
+
// admit what the query does NOT contain, so identity is the wrong
|
|
1338
|
+
// axis.
|
|
1339
|
+
//
|
|
1340
|
+
// WHAT A FUTURE ATTEMPT MUST RESPECT: whatever allocates this privilege
|
|
1341
|
+
// has to (a) keep it available to candidates the query does not contain
|
|
1342
|
+
// — analogies, substitutions, compositions — and (b) not depend on the
|
|
1343
|
+
// estimator's ordering among anchors of equal evidence, because that
|
|
1344
|
+
// ordering is not a fact about the corpus. No lever satisfying both has
|
|
1345
|
+
// been found. Until one is, this premise rests on the enumeration order
|
|
1346
|
+
// recorded above, and test/34 is the only test that notices if it
|
|
1347
|
+
// moves.
|
|
1222
1348
|
dominant = true;
|
|
1223
1349
|
if (pastLeading) {
|
|
1224
1350
|
status = "root";
|
|
@@ -1229,8 +1355,15 @@ export function commitVotes(ctx, pooled, sat, regions, regionVoter, N, td, cfg)
|
|
|
1229
1355
|
}
|
|
1230
1356
|
}
|
|
1231
1357
|
else {
|
|
1232
|
-
|
|
1233
|
-
|
|
1358
|
+
// THE FLOOR AND THE BREAK READ THE IDF WEIGHTING. `floor` is derived
|
|
1359
|
+
// for pooled IDF-weighted votes, and `rootCut` comes from the IDF
|
|
1360
|
+
// distribution (`idfDesc`), so gating the mode-dependent `vote` against
|
|
1361
|
+
// either let a weighting mode change an admission (measured: anchor 87,
|
|
1362
|
+
// inverse 2.682 admitted vs direct 1.468 refused). Reading the IDF sum
|
|
1363
|
+
// makes the verdict mode-independent, and changes nothing in the
|
|
1364
|
+
// engine's own mode, where the two readings coincide.
|
|
1365
|
+
passesNaturalBreak = point.idfVote >= rootCut;
|
|
1366
|
+
passesConsensusFloor = point.idfVote >= floor;
|
|
1234
1367
|
// CO-DOMINANT — an anchor the estimator cannot separate from the
|
|
1235
1368
|
// dominant inherits the dominant's exemption, because that exemption's
|
|
1236
1369
|
// only warrant is being TOP, and "top" is not a fact about the corpus
|
|
@@ -1685,6 +1818,13 @@ ownRootsA, ownRootsB, trace) {
|
|
|
1685
1818
|
outcome,
|
|
1686
1819
|
});
|
|
1687
1820
|
};
|
|
1821
|
+
// ── ADMISSION BY MARGIN, not by votes (see voteRegions' note) ─────────
|
|
1822
|
+
// What this site measures: how far the best ANN proposal's effective score
|
|
1823
|
+
// (score × semanticConfidence) stands above the runner-up's, against
|
|
1824
|
+
// `estimatorNoise(D)`. What it does NOT measure: how many regions voted,
|
|
1825
|
+
// or whether the query's regions agree — that is voteRegions' question, and
|
|
1826
|
+
// here a synthetic gist has already replaced them. The two bars are both
|
|
1827
|
+
// derived (thresholds.md), and neither is a tuning of the other.
|
|
1688
1828
|
let selected = null;
|
|
1689
1829
|
let selectedReach = null;
|
|
1690
1830
|
let selectedIdf = 0;
|
|
@@ -1790,6 +1930,15 @@ async function crossRegionVotes(ctx, query, regions, rvs, k, N, reachMemo, td) {
|
|
|
1790
1930
|
// successfully reconstructed while probing one pair must not be read and
|
|
1791
1931
|
// perceived again while probing another pair in the same climb.
|
|
1792
1932
|
const siblingGistMemo = new Map();
|
|
1933
|
+
// ── ADMISSION BY STRUCTURE, not by a bar (see voteRegions' note) ──────
|
|
1934
|
+
// What this site decides: WHICH regions may pair at all — a region that
|
|
1935
|
+
// already voted (individually discriminative), or a KNOWN non-voting one as
|
|
1936
|
+
// the weak side of a pair whose other side voted; never two non-voting
|
|
1937
|
+
// regions, and never a span contained in a maximal one whose reading is
|
|
1938
|
+
// exact. The bar (the container's idf) comes later, on the candidate. So
|
|
1939
|
+
// its "rejection reasons" name structural disqualifications — a different
|
|
1940
|
+
// vocabulary because it answers a different question, and the three
|
|
1941
|
+
// taxonomies stay separate for the same reason the readings do.
|
|
1793
1942
|
const votedSpans = new Set();
|
|
1794
1943
|
for (const rv of rvs.votes)
|
|
1795
1944
|
votedSpans.add(`${rv.start},${rv.end}`);
|
|
@@ -23,6 +23,22 @@ export declare function leafIdRun(ctx: MindContext, bytes: Uint8Array, from: num
|
|
|
23
23
|
* what a partial prefix means (deposit only interns the whole-stream flat
|
|
24
24
|
* branch when the prefix covers everything). */
|
|
25
25
|
export declare function leafIdPrefix(ctx: MindContext, bytes: Uint8Array): number[];
|
|
26
|
+
/** Which prefixes of `prefix ‖ tail` are STORED NODES — as lengths in the
|
|
27
|
+
* tail's own coordinates, ascending, excluding the empty one. This is the
|
|
28
|
+
* candidate set a rule needs to join an already-stored prefix to a suffix it
|
|
29
|
+
* has not stored: a key names a relation exactly when `prefix ‖ tail[0..p]` IS
|
|
30
|
+
* a node, and a key can end strictly inside the tail without sitting on any
|
|
31
|
+
* fold boundary (a stored member's end is the end of ITS OWN stream, and the
|
|
32
|
+
* fold never emits a cut at a stream's end). Measured: "stockholm mayor"
|
|
33
|
+
* exists, leads on, and its boundary 6 is in neither the tail's cuts nor the
|
|
34
|
+
* concatenation's.
|
|
35
|
+
*
|
|
36
|
+
* ONE cheap content-addressed probe per offset — `leafIdPrefix` walks the bytes
|
|
37
|
+
* once (a point probe each), `findBranch` hashes the growing kid run — and NO
|
|
38
|
+
* `resolve`, which is what keeps this off the O(suffix) vector folds the
|
|
39
|
+
* recognition path pays. It stops at the first byte that was never interned,
|
|
40
|
+
* which costs nothing real: a stored key's bytes are interned by construction. */
|
|
41
|
+
export declare function keyEnds(ctx: MindContext, prefix: Uint8Array, tail: Uint8Array): number[];
|
|
26
42
|
/** The canonical W-window node ids of a byte stream, offset → id — the
|
|
27
43
|
* CONTENT-ADDRESSED IDENTITY of every W-sized slice, under which any content
|
|
28
44
|
* two deposits share IS the same node (hash-consing paid the comparison at
|
|
@@ -70,6 +70,47 @@ export function leafIdPrefix(ctx, bytes) {
|
|
|
70
70
|
}
|
|
71
71
|
return ids;
|
|
72
72
|
}
|
|
73
|
+
/** Which prefixes of `prefix ‖ tail` are STORED NODES — as lengths in the
|
|
74
|
+
* tail's own coordinates, ascending, excluding the empty one. This is the
|
|
75
|
+
* candidate set a rule needs to join an already-stored prefix to a suffix it
|
|
76
|
+
* has not stored: a key names a relation exactly when `prefix ‖ tail[0..p]` IS
|
|
77
|
+
* a node, and a key can end strictly inside the tail without sitting on any
|
|
78
|
+
* fold boundary (a stored member's end is the end of ITS OWN stream, and the
|
|
79
|
+
* fold never emits a cut at a stream's end). Measured: "stockholm mayor"
|
|
80
|
+
* exists, leads on, and its boundary 6 is in neither the tail's cuts nor the
|
|
81
|
+
* concatenation's.
|
|
82
|
+
*
|
|
83
|
+
* ONE cheap content-addressed probe per offset — `leafIdPrefix` walks the bytes
|
|
84
|
+
* once (a point probe each), `findBranch` hashes the growing kid run — and NO
|
|
85
|
+
* `resolve`, which is what keeps this off the O(suffix) vector folds the
|
|
86
|
+
* recognition path pays. It stops at the first byte that was never interned,
|
|
87
|
+
* which costs nothing real: a stored key's bytes are interned by construction. */
|
|
88
|
+
export function keyEnds(ctx, prefix, tail) {
|
|
89
|
+
if (prefix.length === 0 || tail.length === 0)
|
|
90
|
+
return [];
|
|
91
|
+
const joined = new Uint8Array(prefix.length + tail.length);
|
|
92
|
+
joined.set(prefix, 0);
|
|
93
|
+
joined.set(tail, prefix.length);
|
|
94
|
+
const ids = leafIdPrefix(ctx, joined);
|
|
95
|
+
if (ids.length < prefix.length)
|
|
96
|
+
return [];
|
|
97
|
+
const ends = [];
|
|
98
|
+
// The kid run GROWS by one id per offset; `findBranch` wants an array, so the
|
|
99
|
+
// run is built once and pushed into, never re-sliced. Re-slicing
|
|
100
|
+
// `ids.slice(0, prefix.length + p)` per offset made this O(|tail| ·
|
|
101
|
+
// (|prefix| + |tail|)) — quadratic in the tail, where the learning path this
|
|
102
|
+
// follows slices a run that SHRINKS. Same ends, linear copying.
|
|
103
|
+
const run = ids.slice(0, prefix.length);
|
|
104
|
+
// The loop ENDS at the first byte that was never interned (`ids.length`):
|
|
105
|
+
// every later prefix contains it, so none of them can be a node either — this
|
|
106
|
+
// is where the scan stops, not a silent truncation of the answer.
|
|
107
|
+
for (let p = 1; prefix.length + p <= ids.length; p++) {
|
|
108
|
+
run.push(ids[prefix.length + p - 1]);
|
|
109
|
+
if (ctx.store.findBranch(run) !== null)
|
|
110
|
+
ends.push(p);
|
|
111
|
+
}
|
|
112
|
+
return ends;
|
|
113
|
+
}
|
|
73
114
|
/** The canonical W-window node ids of a byte stream, offset → id — the
|
|
74
115
|
* CONTENT-ADDRESSED IDENTITY of every W-sized slice, under which any content
|
|
75
116
|
* two deposits share IS the same node (hash-consing paid the comparison at
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
import type { MindContext } from "./types.js";
|
|
2
|
+
/** One stored experience pair, as bytes. */
|
|
3
|
+
export interface CorpusPair {
|
|
4
|
+
context: Uint8Array;
|
|
5
|
+
continuation: Uint8Array;
|
|
6
|
+
contextId: number;
|
|
7
|
+
continuationId: number;
|
|
8
|
+
/** Bytes of the query this pair was matched on — 0 when browsing. */
|
|
9
|
+
matchedBytes: number;
|
|
10
|
+
/** True when the stored bytes ran past the declared preview capacity. */
|
|
11
|
+
contextTruncated: boolean;
|
|
12
|
+
continuationTruncated: boolean;
|
|
13
|
+
}
|
|
14
|
+
/** Why a search produced no pairs. A STATE, so a caller's own layer can say it
|
|
15
|
+
* in its own words — the byte layer does not speak. */
|
|
16
|
+
export type CorpusMiss = "matched" | "nothing-resolved" | "no-continuations";
|
|
17
|
+
export interface CorpusResult {
|
|
18
|
+
pairs: CorpusPair[];
|
|
19
|
+
/** Query subtrees that content-addressed to a real stored node. */
|
|
20
|
+
resolved: number;
|
|
21
|
+
/** Distinct edge-bearing contexts the climb reached. */
|
|
22
|
+
reached: number;
|
|
23
|
+
/** Distinct contexts that carry a learnt continuation, store-wide. */
|
|
24
|
+
totalContexts: number;
|
|
25
|
+
/** True when these are browse samples rather than search results. */
|
|
26
|
+
browsed: boolean;
|
|
27
|
+
miss: CorpusMiss;
|
|
28
|
+
}
|
|
29
|
+
/** Which stored notes does this query reach? BYTES in, BYTES out.
|
|
30
|
+
*
|
|
31
|
+
* Exact content addressing through the machinery that already exists: the
|
|
32
|
+
* query's recognised sites are the resolved subtrees, the climb goes up from
|
|
33
|
+
* the biggest first, and a pair is a context that carries a continuation. */
|
|
34
|
+
export declare function searchCorpus(ctx: MindContext, queryBytes: Uint8Array, limit?: number): CorpusResult;
|
|
35
|
+
/** Browse real pairs, striding the id space so the sample is spread rather than
|
|
36
|
+
* one local cluster. DETERMINISTIC: `from` is the caller's own offset, so
|
|
37
|
+
* browsing twice with different offsets shows different notes without a random
|
|
38
|
+
* draw (the demo drew `Math.random()`, which the engine cannot do — same seed,
|
|
39
|
+
* same order, same query must mean the same answer). */
|
|
40
|
+
export declare function sampleCorpus(ctx: MindContext, limit?: number, from?: number): CorpusResult;
|
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
// corpus.ts — read the trained memory back out of the DAG, AS DATA.
|
|
2
|
+
//
|
|
3
|
+
// A trained experience pair IS one continuation edge: `src` is the context that
|
|
4
|
+
// was deposited, `dst` is what the mind learnt follows it. Reading them back
|
|
5
|
+
// uses the store's own structure and its own indexes — no auxiliary index is
|
|
6
|
+
// built, nothing is written, and NOTHING HERE KNOWS ABOUT TEXT: this layer takes
|
|
7
|
+
// bytes and returns bytes. The text case is one helper on the Mind
|
|
8
|
+
// (`searchCorpusText`), which encodes, calls this, and decodes.
|
|
9
|
+
//
|
|
10
|
+
// WHERE EACH STAGE COMES FROM (ported from the demo's `explore.ts`, which
|
|
11
|
+
// hand-rolled its own resolution):
|
|
12
|
+
//
|
|
13
|
+
// 1. PERCEIVE, CONTENT-ADDRESS and ADMIT the query — `recognise()`, the SAME
|
|
14
|
+
// machinery an answer goes through. It returns the sites: the query spans
|
|
15
|
+
// that content-addressed to a stored node that can lead somewhere. The
|
|
16
|
+
// demo's own recursive `findLeaf`/`findBranch` walk was a second
|
|
17
|
+
// implementation of exactly this, and it is NOT ported.
|
|
18
|
+
// 2. CLIMB the structural `kid` table from each resolved site to the
|
|
19
|
+
// edge-bearing contexts above it (`edgeAncestors`, traverse.ts), weighting
|
|
20
|
+
// each context by how much query content reached it.
|
|
21
|
+
// 3. READ the continuation off the edge table (`nextFirst`).
|
|
22
|
+
//
|
|
23
|
+
// COST is set by how much of the QUERY resolves, never by the size of the
|
|
24
|
+
// store: the sites are what recognition already found, the climb is bounded by
|
|
25
|
+
// the declared `corpusClimbs`/`corpusContextsPerClimb`, and every store call is
|
|
26
|
+
// a point probe or a capped read. All work is accounted by the store's own
|
|
27
|
+
// meter hooks when a response's meter is open — there is no second instrument.
|
|
28
|
+
//
|
|
29
|
+
// WHAT THIS IS NOT. Exact content addressing, not fuzzy keyword search: a query
|
|
30
|
+
// shares results with a stored note when it shares actual chunk-aligned content
|
|
31
|
+
// with it. An arbitrary mid-word fragment resolves to nothing, and the honest
|
|
32
|
+
// answer there is "nothing matched" — which is why `sampleCorpus` exists, and
|
|
33
|
+
// why the miss is reported as a STATE rather than as prose (the text helper
|
|
34
|
+
// turns it into words).
|
|
35
|
+
import { recognise } from "./recognition.js";
|
|
36
|
+
import { edgeAncestors } from "./traverse.js";
|
|
37
|
+
/** A context node as a pair, or null when it carries no continuation. */
|
|
38
|
+
function pairOf(ctx, id, matchedBytes) {
|
|
39
|
+
const outs = ctx.store.nextFirst(id, 1);
|
|
40
|
+
if (outs.length === 0)
|
|
41
|
+
return null;
|
|
42
|
+
const cap = ctx.cfg.corpusPreviewBytes;
|
|
43
|
+
const context = ctx.store.bytesPrefix(id, cap + 1);
|
|
44
|
+
const continuation = ctx.store.bytesPrefix(outs[0], cap + 1);
|
|
45
|
+
const contextTruncated = context.length > cap;
|
|
46
|
+
const continuationTruncated = continuation.length > cap;
|
|
47
|
+
return {
|
|
48
|
+
context: contextTruncated ? context.subarray(0, cap) : context,
|
|
49
|
+
continuation: continuationTruncated
|
|
50
|
+
? continuation.subarray(0, cap)
|
|
51
|
+
: continuation,
|
|
52
|
+
contextId: id,
|
|
53
|
+
continuationId: outs[0],
|
|
54
|
+
matchedBytes,
|
|
55
|
+
contextTruncated,
|
|
56
|
+
continuationTruncated,
|
|
57
|
+
};
|
|
58
|
+
}
|
|
59
|
+
/** Which stored notes does this query reach? BYTES in, BYTES out.
|
|
60
|
+
*
|
|
61
|
+
* Exact content addressing through the machinery that already exists: the
|
|
62
|
+
* query's recognised sites are the resolved subtrees, the climb goes up from
|
|
63
|
+
* the biggest first, and a pair is a context that carries a continuation. */
|
|
64
|
+
export function searchCorpus(ctx, queryBytes, limit) {
|
|
65
|
+
const store = ctx.store;
|
|
66
|
+
const want = Math.max(1, Math.min(limit ?? ctx.cfg.corpusContextsPerClimb, ctx.cfg.corpusLimitMax));
|
|
67
|
+
// A resolved subtree must account for at least one window (W): a single
|
|
68
|
+
// character resolves against almost any store and means nothing. W is the
|
|
69
|
+
// mind's own line between chance and evidence — derived, never declared.
|
|
70
|
+
const floor = ctx.space.maxGroup;
|
|
71
|
+
const resolved = recognise(ctx, queryBytes).sites
|
|
72
|
+
.map((s) => ({ id: s.payload, len: store.contentLen(s.payload, 512) }))
|
|
73
|
+
.filter((r) => r.len >= floor);
|
|
74
|
+
// Biggest first, lowest id breaking ties: a clause is evidence, a character is
|
|
75
|
+
// noise, and equal evidence must not be decided by iteration order.
|
|
76
|
+
const byLength = resolved
|
|
77
|
+
.sort((a, b) => b.len - a.len || a.id - b.id)
|
|
78
|
+
.slice(0, ctx.cfg.corpusClimbs);
|
|
79
|
+
// Weight each context by how much query content reached it.
|
|
80
|
+
const weight = new Map();
|
|
81
|
+
for (const { id, len } of byLength) {
|
|
82
|
+
for (const root of edgeAncestors(ctx, id, ctx.cfg.corpusContextsPerClimb).roots) {
|
|
83
|
+
weight.set(root, (weight.get(root) ?? 0) + len);
|
|
84
|
+
}
|
|
85
|
+
}
|
|
86
|
+
const pairs = [];
|
|
87
|
+
const ranked = [...weight.entries()].sort((a, b) => b[1] - a[1] || a[0] - b[0]);
|
|
88
|
+
for (const [id, w] of ranked) {
|
|
89
|
+
if (pairs.length >= want)
|
|
90
|
+
break;
|
|
91
|
+
const pair = pairOf(ctx, id, w);
|
|
92
|
+
if (pair)
|
|
93
|
+
pairs.push(pair);
|
|
94
|
+
}
|
|
95
|
+
return {
|
|
96
|
+
pairs,
|
|
97
|
+
resolved: byLength.length,
|
|
98
|
+
reached: weight.size,
|
|
99
|
+
totalContexts: store.edgeSourceCount(),
|
|
100
|
+
browsed: false,
|
|
101
|
+
miss: pairs.length > 0
|
|
102
|
+
? "matched"
|
|
103
|
+
: weight.size === 0
|
|
104
|
+
? "nothing-resolved"
|
|
105
|
+
: "no-continuations",
|
|
106
|
+
};
|
|
107
|
+
}
|
|
108
|
+
/** Browse real pairs, striding the id space so the sample is spread rather than
|
|
109
|
+
* one local cluster. DETERMINISTIC: `from` is the caller's own offset, so
|
|
110
|
+
* browsing twice with different offsets shows different notes without a random
|
|
111
|
+
* draw (the demo drew `Math.random()`, which the engine cannot do — same seed,
|
|
112
|
+
* same order, same query must mean the same answer). */
|
|
113
|
+
export function sampleCorpus(ctx, limit, from = 0) {
|
|
114
|
+
const store = ctx.store;
|
|
115
|
+
const want = Math.max(1, Math.min(limit ?? ctx.cfg.corpusContextsPerClimb, ctx.cfg.corpusLimitMax));
|
|
116
|
+
const total = store.nodeCount();
|
|
117
|
+
const probes = ctx.cfg.corpusSampleProbes;
|
|
118
|
+
const floorBytes = ctx.cfg.corpusSampleFloorBytes;
|
|
119
|
+
const pairs = [];
|
|
120
|
+
// EACH CONTEXT AT MOST ONCE. Striding the id space revisits ids when the
|
|
121
|
+
// store is small relative to the probe budget (measured: a 160-node store
|
|
122
|
+
// returned the SAME pair six times for `limit: 6`), and a browse that repeats
|
|
123
|
+
// itself is not a browse. The demo had the same hole; it is invisible only on
|
|
124
|
+
// a store far larger than the probe budget.
|
|
125
|
+
const seen = new Set();
|
|
126
|
+
for (let i = 0; i < probes && pairs.length < want && total > 0; i++) {
|
|
127
|
+
const slot = (i / probes + from) % 1;
|
|
128
|
+
const id = Math.floor(slot * total);
|
|
129
|
+
if (seen.has(id))
|
|
130
|
+
continue;
|
|
131
|
+
if (!store.has(id) || !store.hasNext(id))
|
|
132
|
+
continue;
|
|
133
|
+
if (store.contentLen(id, floorBytes) < floorBytes)
|
|
134
|
+
continue;
|
|
135
|
+
const pair = pairOf(ctx, id, 0);
|
|
136
|
+
if (pair) {
|
|
137
|
+
seen.add(id);
|
|
138
|
+
pairs.push(pair);
|
|
139
|
+
}
|
|
140
|
+
}
|
|
141
|
+
return {
|
|
142
|
+
pairs,
|
|
143
|
+
resolved: 0,
|
|
144
|
+
reached: pairs.length,
|
|
145
|
+
totalContexts: store.edgeSourceCount(),
|
|
146
|
+
browsed: true,
|
|
147
|
+
miss: pairs.length > 0 ? "matched" : "no-continuations",
|
|
148
|
+
};
|
|
149
|
+
}
|
|
@@ -162,6 +162,13 @@ export declare class GraphSearch {
|
|
|
162
162
|
* recursive completion), and chooseNext (distributional-evidence edge
|
|
163
163
|
* disambiguation when a recognised form has multiple continuations). */
|
|
164
164
|
host: GraphSearchHost);
|
|
165
|
+
/** The nodes the QUERY canonically names — the same identity the store's keys
|
|
166
|
+
* were written through. A byte-exact test is not enough: the query writes
|
|
167
|
+
* `Eiffel Tower country` and the deposited node is `eiffel tower country`, so
|
|
168
|
+
* a join that filters the query's own subject by RAW bytes re-admits it —
|
|
169
|
+
* measured: that is the trap's wrong answer (`The capital of Eiffel Tower
|
|
170
|
+
* country is Berlin.`). Cached by query identity, because the search is
|
|
171
|
+
* reused across responses. */
|
|
165
172
|
private hubBound;
|
|
166
173
|
/** Explore the Sema graph for the lightest cover of the query and return its
|
|
167
174
|
* chosen spans left-to-right — WITH the derivation's total weight (the g
|