@hviana/sema 0.8.1 → 0.8.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +29 -29
- package/TRADEMARKS.md +0 -1
- package/dist/src/config.d.ts +28 -0
- package/dist/src/config.js +20 -0
- package/dist/src/geometry.d.ts +21 -0
- package/dist/src/geometry.js +21 -0
- package/dist/src/meter.d.ts +76 -0
- package/dist/src/meter.js +95 -0
- package/dist/src/mind/attention.d.ts +4 -0
- package/dist/src/mind/attention.js +165 -16
- package/dist/src/mind/canonical.d.ts +16 -0
- package/dist/src/mind/canonical.js +41 -0
- package/dist/src/mind/corpus.d.ts +40 -0
- package/dist/src/mind/corpus.js +149 -0
- package/dist/src/mind/graph-search.d.ts +7 -0
- package/dist/src/mind/graph-search.js +254 -24
- package/dist/src/mind/index.d.ts +3 -1
- package/dist/src/mind/index.js +1 -0
- package/dist/src/mind/match.d.ts +9 -4
- package/dist/src/mind/match.js +147 -61
- package/dist/src/mind/mechanisms/cast.js +19 -3
- package/dist/src/mind/mechanisms/confluence.js +24 -0
- package/dist/src/mind/mechanisms/cover.js +6 -0
- package/dist/src/mind/mechanisms/recall.js +32 -4
- package/dist/src/mind/mind.d.ts +57 -0
- package/dist/src/mind/mind.js +72 -1
- package/dist/src/mind/pipeline-mechanism.d.ts +7 -0
- package/dist/src/mind/pipeline.js +66 -20
- package/dist/src/mind/primitives.js +9 -1
- package/dist/src/mind/rationale.d.ts +28 -1
- package/dist/src/mind/rationale.js +22 -1
- package/dist/src/mind/reasoning.d.ts +25 -3
- package/dist/src/mind/reasoning.js +125 -20
- package/dist/src/mind/recognition.js +4 -8
- package/dist/src/mind/resonance.js +20 -1
- package/dist/src/mind/trace.js +1 -0
- package/dist/src/mind/traverse.js +15 -3
- package/dist/src/mind/types.d.ts +49 -4
- package/docs/INVARIANTS.md +2 -2
- package/docs/architecture/bounded-reads.md +1 -1
- package/docs/architecture/commonality.md +2 -2
- package/docs/architecture/cost-model.md +2 -2
- package/docs/architecture/determinism.md +7 -7
- package/docs/architecture/match-project.md +2 -3
- package/docs/architecture/mechanism-market.md +10 -10
- package/docs/architecture/meter.md +5 -5
- package/docs/architecture/store.md +3 -3
- package/docs/failures/tempting-but-wrong.md +34 -6
- package/docs/harness/gates.md +2 -2
- package/docs/mechanisms/cast.md +2 -2
- package/docs/mechanisms/cover.md +2 -3
- package/docs/mechanisms/extraction.md +7 -7
- package/docs/mechanisms/recall.md +8 -9
- package/jsr.json +1 -1
- package/package.json +1 -1
- package/src/alu/README.md +11 -12
- package/src/config.ts +48 -0
- package/src/geometry.ts +21 -0
- package/src/meter.ts +98 -0
- package/src/mind/attention.ts +167 -16
- package/src/mind/canonical.ts +43 -0
- package/src/mind/corpus.ts +202 -0
- package/src/mind/graph-search.ts +277 -23
- package/src/mind/index.ts +8 -1
- package/src/mind/match.ts +148 -57
- package/src/mind/mechanisms/cast.ts +20 -2
- package/src/mind/mechanisms/confluence.ts +24 -0
- package/src/mind/mechanisms/cover.ts +5 -0
- package/src/mind/mechanisms/recall.ts +32 -4
- package/src/mind/mind.ts +125 -0
- package/src/mind/pipeline-mechanism.ts +7 -0
- package/src/mind/pipeline.ts +79 -22
- package/src/mind/primitives.ts +9 -1
- package/src/mind/rationale.ts +35 -1
- package/src/mind/reasoning.ts +145 -13
- package/src/mind/recognition.ts +4 -8
- package/src/mind/resonance.ts +19 -1
- package/src/mind/trace.ts +1 -0
- package/src/mind/traverse.ts +16 -6
- package/src/mind/types.ts +53 -4
- package/test/100-complete-grounding-trace.test.mjs +109 -0
- package/test/101-alignment-gap-bound.test.mjs +106 -0
- package/test/102-production-composes-at-scale.test.mjs +110 -0
- package/test/103-alignment-gap-budget.test.mjs +89 -0
- package/test/104-composition-is-reported.test.mjs +90 -0
- package/test/105-derive-through-reports-its-refusal.test.mjs +137 -0
- package/test/106-the-join-fires.test.mjs +94 -0
- package/test/107-the-join-is-counted.test.mjs +81 -0
- package/test/108-the-join-chains.test.mjs +78 -0
- package/test/109-the-pivot-is-counted.test.mjs +60 -0
- package/test/110-the-reasoner-stops-when-the-question-is-answered.test.mjs +91 -0
- package/test/111-the-cover-assembly-is-counted.test.mjs +74 -0
- package/test/112-the-exploration-does-not-grow-with-the-hub.test.mjs +89 -0
- package/test/113-the-rationale-payload-is-bounded.test.mjs +84 -0
- package/test/114-alignment-budget-is-per-sweep.test.mjs +93 -0
- package/test/116-the-extension-is-gated-by-the-pipelines-own-remainder.test.mjs +100 -0
- package/test/117-corpus-search.test.mjs +171 -0
- package/test/118-the-join-reaches-a-key-off-the-cut.test.mjs +74 -0
- package/test/119-the-work-does-not-grow-with-the-corpus.test.mjs +122 -0
- package/test/120-composition-is-consequence.test.mjs +132 -0
- package/test/121-the-extension-does-not-grow-with-the-corpus.test.mjs +128 -0
- package/test/122-the-climb-search-does-not-grow-with-the-corpus.test.mjs +117 -0
- package/test/123-the-paired-formulas-agree.test.mjs +90 -0
- package/test/125-the-post-grounding-branch-publishes-its-operand.test.mjs +51 -0
- package/test/126-the-pipeline-does-not-name-mechanisms.test.mjs +42 -0
- package/test/128-the-leads-somewhere-pair-agrees.test.mjs +83 -0
- package/test/129-the-trace-payload-shape.test.mjs +164 -0
- package/test/14-scaling.test.mjs +10 -7
- package/test/32-confluence.test.mjs +68 -0
- package/test/38-reason-restate-guard.test.mjs +8 -2
- package/test/43-cast-analog-seat.test.mjs +10 -0
- package/test/55-cost-meter.test.mjs +859 -0
- package/test/76-reference-binding.test.mjs +6 -1
- package/test/89-completion-recursion.test.mjs +30 -5
package/src/mind/match.ts
CHANGED
|
@@ -38,7 +38,7 @@ import {
|
|
|
38
38
|
identityBar,
|
|
39
39
|
significanceBar,
|
|
40
40
|
} from "../geometry.js";
|
|
41
|
-
import { bytesEqual, indexOf } from "../bytes.js";
|
|
41
|
+
import { bytesEqual, indexOf, latin1 } from "../bytes.js";
|
|
42
42
|
import type { MindContext } from "./types.js";
|
|
43
43
|
import { chainReach, leafIdRun } from "./canonical.js";
|
|
44
44
|
import { foldTree, gistOf, perceive, read, resolve } from "./primitives.js";
|
|
@@ -54,6 +54,7 @@ import {
|
|
|
54
54
|
sharedReachMemo,
|
|
55
55
|
} from "./traverse.js";
|
|
56
56
|
import { recognise, segment } from "./recognition.js";
|
|
57
|
+
import { rItem } from "./trace.js";
|
|
57
58
|
import type { Site } from "./graph-search.js";
|
|
58
59
|
|
|
59
60
|
// ═══════════════════════════════════════════════════════════════════════════
|
|
@@ -360,9 +361,14 @@ export interface AlignGap {
|
|
|
360
361
|
|
|
361
362
|
/** Extend a seed match (query offset qo ↔ candidate offset co) to its maximal
|
|
362
363
|
* common run, then walk outward in both directions collecting further common
|
|
363
|
-
* runs of at least W bytes across
|
|
364
|
-
*
|
|
365
|
-
*
|
|
364
|
+
* runs of at least W bytes across mismatch gaps. Each gap's LENGTH is the
|
|
365
|
+
* pair's own extent (a gap cannot be longer than the bytes it spans) and the
|
|
366
|
+
* sweep's WORK is proportional to the bytes a run spans (the context's windows
|
|
367
|
+
* are indexed once, then the query's are walked) — the arity bound
|
|
368
|
+
* (`chainReach`) used to cap BOTH, and truncated every learned frame whose
|
|
369
|
+
* slot was longer. Each sweep owns its own budget, so an exhausted right
|
|
370
|
+
* sweep never starves the left one. Returns the matched query spans and the
|
|
371
|
+
* mismatch pairs between consecutive runs.
|
|
366
372
|
*
|
|
367
373
|
* This is the SEEDED aligner, distinct from {@link alignRuns}: that one finds
|
|
368
374
|
* every run two structures share anywhere (a weave), this one reads two
|
|
@@ -382,7 +388,21 @@ export function alignAround(
|
|
|
382
388
|
co: number,
|
|
383
389
|
): { matched: Array<[number, number]>; gaps: AlignGap[] } {
|
|
384
390
|
const W = ctx.space.maxGroup;
|
|
385
|
-
|
|
391
|
+
// THE GAP LENGTH IS THE PAIR'S OWN EXTENT; THE WORK IS BUDGETED.
|
|
392
|
+
//
|
|
393
|
+
// The sweep walks (queryGap, contextGap) pairs by ASCENDING total, so reaching
|
|
394
|
+
// a gap of size G costs about G²/2 pairs. Bounding the LENGTH by the write
|
|
395
|
+
// side's arity (`chainReach(W)` = 16) therefore truncated every learned frame
|
|
396
|
+
// whose slot is longer — measured: `bindReference` reported the cap at 18, 24,
|
|
397
|
+
// 30 and 36 bytes and `recall` answered with ANOTHER instance's filler — while
|
|
398
|
+
// removing the bound outright took the corpus-cost guard (test/89) from
|
|
399
|
+
// milliseconds to 68 seconds.
|
|
400
|
+
//
|
|
401
|
+
// Bounding the PAIRS keeps a call's cost constant however long the pair is,
|
|
402
|
+
// and the ascending order means an exhausted budget drops the FAR
|
|
403
|
+
// continuations and never the near ones — the same degradation recognition.ts
|
|
404
|
+
// documents for its canon budget. Length and work are different questions;
|
|
405
|
+
// this is the one place they were conflated.
|
|
386
406
|
// Maximal run around the seed.
|
|
387
407
|
let qs = qo, ss = co;
|
|
388
408
|
while (qs > 0 && ss > 0 && q[qs - 1] === c[ss - 1]) {
|
|
@@ -396,8 +416,60 @@ export function alignAround(
|
|
|
396
416
|
}
|
|
397
417
|
const matched: Array<[number, number]> = [[qs, qe]];
|
|
398
418
|
const gaps: AlignGap[] = [];
|
|
399
|
-
//
|
|
400
|
-
//
|
|
419
|
+
// THE SWEEP IS STRUCTURAL, NOT ENUMERATIVE.
|
|
420
|
+
//
|
|
421
|
+
// The criterion is unchanged: the next common run, MINIMUM TOTAL GAP, ties to
|
|
422
|
+
// the smaller query gap. What changed is how it is found. Enumerating
|
|
423
|
+
// (queryGap, contextGap) pairs by ascending total reaches a run at total t in
|
|
424
|
+
// about t²/2 pairs — and that quadratic shape, not the reach, was the cost
|
|
425
|
+
// problem: capping the pairs dropped reach (a legitimate 24-byte slot stopped
|
|
426
|
+
// being found), while leaving them uncapped cost 68 seconds on the corpus
|
|
427
|
+
// guard. Neither is the answer, because the answer is the algorithm.
|
|
428
|
+
//
|
|
429
|
+
// The context's windows are indexed ONCE, for lengths 1..W — W being the
|
|
430
|
+
// geometry's own unit of composition, so nothing is chosen here. Each step
|
|
431
|
+
// then walks the query's windows outward from the anchor: for a given query
|
|
432
|
+
// gap the nearest context gap that continues a run is one O(1) lookup, and the
|
|
433
|
+
// walk stops the moment the query gap alone exceeds the best total already
|
|
434
|
+
// found. So the work is proportional to the bytes the run SPANS. No budget,
|
|
435
|
+
// no cap, no number: a long slot is reached, and its price is already the
|
|
436
|
+
// ladder's (its bytes are unaccounted, so the search pays PASS per byte).
|
|
437
|
+
const index: Array<Map<string, number[]>> = [];
|
|
438
|
+
for (let len = 1; len <= W; len++) {
|
|
439
|
+
const m = new Map<string, number[]>();
|
|
440
|
+
for (let o = 0; o + len <= c.length; o++) {
|
|
441
|
+
const key = latin1(c.subarray(o, o + len));
|
|
442
|
+
const at = m.get(key);
|
|
443
|
+
if (at === undefined) m.set(key, [o]);
|
|
444
|
+
else at.push(o);
|
|
445
|
+
}
|
|
446
|
+
index.push(m);
|
|
447
|
+
}
|
|
448
|
+
/** Smallest listed offset at or after `from`, or -1. */
|
|
449
|
+
const fromAt = (list: number[], from: number): number => {
|
|
450
|
+
let lo = 0, hi = list.length - 1, best = -1;
|
|
451
|
+
while (lo <= hi) {
|
|
452
|
+
const mid = (lo + hi) >> 1;
|
|
453
|
+
if (list[mid] >= from) {
|
|
454
|
+
best = list[mid];
|
|
455
|
+
hi = mid - 1;
|
|
456
|
+
} else lo = mid + 1;
|
|
457
|
+
}
|
|
458
|
+
return best;
|
|
459
|
+
};
|
|
460
|
+
/** Largest listed offset at or before `to`, or -1. */
|
|
461
|
+
const toAt = (list: number[], to: number): number => {
|
|
462
|
+
let lo = 0, hi = list.length - 1, best = -1;
|
|
463
|
+
while (lo <= hi) {
|
|
464
|
+
const mid = (lo + hi) >> 1;
|
|
465
|
+
if (list[mid] <= to) {
|
|
466
|
+
best = list[mid];
|
|
467
|
+
lo = mid + 1;
|
|
468
|
+
} else hi = mid - 1;
|
|
469
|
+
}
|
|
470
|
+
return best;
|
|
471
|
+
};
|
|
472
|
+
/** Length of the common run STARTING at (qi, si). */
|
|
401
473
|
const runLenAt = (qi: number, si: number): number => {
|
|
402
474
|
let n = 0;
|
|
403
475
|
while (qi + n < q.length && si + n < c.length && q[qi + n] === c[si + n]) {
|
|
@@ -405,63 +477,80 @@ export function alignAround(
|
|
|
405
477
|
}
|
|
406
478
|
return n;
|
|
407
479
|
};
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
|
|
421
|
-
|
|
480
|
+
/** Length of the common run ENDING at (qi, si). */
|
|
481
|
+
const runLenBefore = (qi: number, si: number): number => {
|
|
482
|
+
let n = 0;
|
|
483
|
+
while (n < qi && n < si && q[qi - 1 - n] === c[si - 1 - n]) n++;
|
|
484
|
+
return n;
|
|
485
|
+
};
|
|
486
|
+
/** The next run outward from an anchor, or null when the bytes run out. */
|
|
487
|
+
const nextRun = (
|
|
488
|
+
qi: number,
|
|
489
|
+
si: number,
|
|
490
|
+
forward: boolean,
|
|
491
|
+
): { gq: number; gs: number; n: number } | null => {
|
|
492
|
+
const qLim = forward ? q.length - qi : qi;
|
|
493
|
+
let best: { gq: number; gs: number; n: number } | null = null;
|
|
494
|
+
for (let gq = 0; gq < qLim; gq++) {
|
|
495
|
+
// No later query gap can beat a total already found.
|
|
496
|
+
if (best !== null && gq > best.gq + best.gs) break;
|
|
497
|
+
const left = qLim - gq;
|
|
498
|
+
// A run of >= W bytes, or — when the query itself ends inside one window —
|
|
499
|
+
// the run that REACHES that end. Exactly the acceptance the sweep had.
|
|
500
|
+
const lens = left >= W ? [W] : [left];
|
|
501
|
+
for (const len of lens) {
|
|
502
|
+
const key = latin1(
|
|
503
|
+
q.subarray(
|
|
504
|
+
forward ? qi + gq : qi - gq - len,
|
|
505
|
+
forward ? qi + gq + len : qi - gq,
|
|
506
|
+
),
|
|
507
|
+
);
|
|
508
|
+
const list = index[len - 1].get(key);
|
|
509
|
+
if (list === undefined) continue;
|
|
510
|
+
const o = forward ? fromAt(list, si) : toAt(list, si - len);
|
|
511
|
+
if (o < 0) continue;
|
|
512
|
+
const n = forward
|
|
513
|
+
? runLenAt(qi + gq, o)
|
|
514
|
+
: runLenBefore(qi - gq, o + len);
|
|
515
|
+
if (n < 1) continue;
|
|
516
|
+
if (
|
|
517
|
+
forward ? n >= W || qi + gq + n === q.length : n >= W || n === qi - gq
|
|
518
|
+
) {
|
|
519
|
+
const gs = forward ? o - si : si - len - o;
|
|
520
|
+
if (best === null || gq + gs < best.gq + best.gs) {
|
|
521
|
+
best = { gq, gs, n };
|
|
422
522
|
}
|
|
423
|
-
matched.push([qi + gq, qi + gq + n]);
|
|
424
|
-
qi = qi + gq + n;
|
|
425
|
-
si = si + gs + n;
|
|
426
|
-
found = true;
|
|
427
523
|
break;
|
|
428
524
|
}
|
|
429
525
|
}
|
|
430
526
|
}
|
|
431
|
-
|
|
527
|
+
return best;
|
|
528
|
+
};
|
|
529
|
+
// RIGHT sweep.
|
|
530
|
+
let qi = qe, si = se;
|
|
531
|
+
for (;;) {
|
|
532
|
+
const step = nextRun(qi, si, true);
|
|
533
|
+
if (step === null) break;
|
|
534
|
+
if (step.gq > 0 || step.gs > 0) {
|
|
535
|
+
gaps.push({ qs: qi, qe: qi + step.gq, cs: si, ce: si + step.gs });
|
|
536
|
+
}
|
|
537
|
+
matched.push([qi + step.gq, qi + step.gq + step.n]);
|
|
538
|
+
qi = qi + step.gq + step.n;
|
|
539
|
+
si = si + step.gs + step.n;
|
|
432
540
|
}
|
|
433
|
-
// LEFT sweep (mirror)
|
|
541
|
+
// LEFT sweep (mirror): an independent walk, so an exhausted right side can
|
|
542
|
+
// never starve it (pinned by test/114).
|
|
434
543
|
qi = qs;
|
|
435
544
|
si = ss;
|
|
436
545
|
for (;;) {
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
|
|
441
|
-
if (gs > reachCap) continue;
|
|
442
|
-
if (qi - gq <= 0 || si - gs <= 0) continue;
|
|
443
|
-
// Run ENDING at (qi - gq, si - gs).
|
|
444
|
-
let n = 0;
|
|
445
|
-
while (
|
|
446
|
-
n < qi - gq && n < si - gs &&
|
|
447
|
-
q[qi - gq - 1 - n] === c[si - gs - 1 - n]
|
|
448
|
-
) {
|
|
449
|
-
n++;
|
|
450
|
-
}
|
|
451
|
-
if (n >= W || n === qi - gq) {
|
|
452
|
-
if (n === 0) continue;
|
|
453
|
-
if (gq > 0 || gs > 0) {
|
|
454
|
-
gaps.push({ qs: qi - gq, qe: qi, cs: si - gs, ce: si });
|
|
455
|
-
}
|
|
456
|
-
matched.push([qi - gq - n, qi - gq]);
|
|
457
|
-
qi = qi - gq - n;
|
|
458
|
-
si = si - gs - n;
|
|
459
|
-
found = true;
|
|
460
|
-
break;
|
|
461
|
-
}
|
|
462
|
-
}
|
|
546
|
+
const step = nextRun(qi, si, false);
|
|
547
|
+
if (step === null) break;
|
|
548
|
+
if (step.gq > 0 || step.gs > 0) {
|
|
549
|
+
gaps.push({ qs: qi - step.gq, qe: qi, cs: si - step.gs, ce: si });
|
|
463
550
|
}
|
|
464
|
-
|
|
551
|
+
matched.push([qi - step.gq - step.n, qi - step.gq]);
|
|
552
|
+
qi = qi - step.gq - step.n;
|
|
553
|
+
si = si - step.gs - step.n;
|
|
465
554
|
}
|
|
466
555
|
return { matched, gaps };
|
|
467
556
|
}
|
|
@@ -634,7 +723,7 @@ export function frameSlots(
|
|
|
634
723
|
* ANCHOR that the query displaced. Neither implies the other, and the
|
|
635
724
|
* observed failures pass the restatement guard cleanly.
|
|
636
725
|
*
|
|
637
|
-
*
|
|
726
|
+
* Four conditions, all byte-exact and all necessary:
|
|
638
727
|
*
|
|
639
728
|
* 1. the query and the anchor must be ONE STRUCTURE — what they share has to
|
|
640
729
|
* dominate the query, or the query is not a variant of the anchor at all
|
|
@@ -722,8 +811,10 @@ export function substituteAll(
|
|
|
722
811
|
const usable = pairs.filter((p) => p.needle.length > 0);
|
|
723
812
|
if (usable.length === 0) return hay;
|
|
724
813
|
// Longest needle first, so a needle that is a prefix of another can never
|
|
725
|
-
// pre-empt it. Ties cannot arise:
|
|
726
|
-
// pairwise distinct
|
|
814
|
+
// pre-empt it. Ties cannot arise: a consumer that VOICES checks the
|
|
815
|
+
// fillers pairwise with `distinct` and refuses such an instance itself —
|
|
816
|
+
// `frameSlots` reports and does not judge (see its own doc), so the refusal
|
|
817
|
+
// lives with the mechanism that needs it, not here.
|
|
727
818
|
const order = [...usable].sort((a, b) => b.needle.length - a.needle.length);
|
|
728
819
|
const out: number[] = [];
|
|
729
820
|
let i = 0;
|
|
@@ -615,7 +615,23 @@ export async function counterfactualTransfer(
|
|
|
615
615
|
fwd !== null && indexOf(answer, fwd, 0) < 0 &&
|
|
616
616
|
!restatesQuery(query, fwd)
|
|
617
617
|
) {
|
|
618
|
-
|
|
618
|
+
// THROUGH THE SHARED JOINER, not a bare concatenation.
|
|
619
|
+
//
|
|
620
|
+
// `joinWithBridge` is the composition step every out-of-search assembly
|
|
621
|
+
// shares (multi-topic fusion, CAST's substitution and comparison): it
|
|
622
|
+
// asks the corpus for a learnt connector between the pieces and, on a
|
|
623
|
+
// miss, joins them BARE **and says so** — the `bridgeMiss` step (see
|
|
624
|
+
// resonance.ts). This site bypassed it, and that is the whole of the
|
|
625
|
+
// gluing the study measured: `"Steel is hard"` + `"wet"` came back as
|
|
626
|
+
// `"hardwet"`, `"eva director father"` + `"The father of…"` as
|
|
627
|
+
// `"fatherThe"` — compositions no rationale could show, because the one
|
|
628
|
+
// step that made them left no trace.
|
|
629
|
+
//
|
|
630
|
+
// Routing it through the shared joiner is the instrumentation fix that
|
|
631
|
+
// comes first: a bare join stays possible (the house rule is "joined
|
|
632
|
+
// bare, never silent") but it is now VISIBLE, and an attested connector
|
|
633
|
+
// is used when the corpus has one.
|
|
634
|
+
answer = await joinWithBridge(ctx, answer, fwd);
|
|
619
635
|
}
|
|
620
636
|
ctx.trace?.step(
|
|
621
637
|
"projectCounterfactual",
|
|
@@ -851,7 +867,9 @@ export async function counterfactualTransfer(
|
|
|
851
867
|
// grounds") — fine for ORIENTING mechanisms, not for voicing learnt
|
|
852
868
|
// content the query never asked about. Computed once here; both the
|
|
853
869
|
// hub fallback below and the comparison gate consume it.
|
|
854
|
-
const rootTrusted = roots.some((r) =>
|
|
870
|
+
const rootTrusted = roots.some((r) =>
|
|
871
|
+
r.idfVote >= consensusFloor(corpusN(ctx))
|
|
872
|
+
); // the IDF sum: the bar's own quantity
|
|
855
873
|
// The context that ESTABLISHES a filler — the same reverse context, under
|
|
856
874
|
// the same naming test, `seatOfNode` uses to VOICE an analog (a predecessor
|
|
857
875
|
// whose bytes CONTAIN the node's: it names or describes it, rather than
|
|
@@ -146,6 +146,30 @@ export async function confluenceJoin(
|
|
|
146
146
|
const bindsAConstituent = (cover: Array<[number, number]>): boolean =>
|
|
147
147
|
cover.some(([cs, ce]) => ce - cs >= 2 * W);
|
|
148
148
|
|
|
149
|
+
// THE VOTE ENTERS AS ORDER, NEVER AS A BAR. This is the only one of the
|
|
150
|
+
// climb's four consumers (recall, fuseAttention, cast, here) that uses the
|
|
151
|
+
// evidence's MAGNITUDE without a floor, and it is legitimate by construction:
|
|
152
|
+
// `ranked` answers "which anchor is stronger" — a question about votes, so the
|
|
153
|
+
// comparison stays within one dimension — and the vote is otherwise only
|
|
154
|
+
// REPORTED (Stream.vote travels to the rationale's constraint nodes). What
|
|
155
|
+
// actually SELECTS a constraint is byte-structural and never the magnitude: a
|
|
156
|
+
// run of at least 2W (`bindsAConstituent`, with its accidental-sharing
|
|
157
|
+
// counter-examples above), disjoint covers (`disjoint`), and scaffolding never
|
|
158
|
+
// binds at all (`dominates(reachOf(…), N)`). The MEET such a stream may
|
|
159
|
+
// produce is selected the same way: a span shorter than 2W is rejected, and
|
|
160
|
+
// the winner is the one with the smallest `reach` (ties broken by the longer
|
|
161
|
+
// span) — a corpus quantity and bytes, never the vote, which appears only in
|
|
162
|
+
// the trace item.
|
|
163
|
+
// binds at all (`dominates(reachOf(…), N)`). The only cut in this loop is a
|
|
164
|
+
// BUDGET, and it is measured: stopping the scan at 2W anchors saves 50-70% of
|
|
165
|
+
// confluence's cost on non-conjunctive queries while preserving every genuinely
|
|
166
|
+
// conjunctive case, whose top anchors ARE its constraints.
|
|
167
|
+
// MEASURED (this goal, on THIS file's own conjunctive fixture): the two
|
|
168
|
+
// streams appear at ranks 1 and 4 against a budget of 2W = 8, on a query whose
|
|
169
|
+
// `ranked` is 9 — so the cut IS live (it would have returned null at the 8th
|
|
170
|
+
// anchor) and it does NOT prune the case it exists to protect. The other
|
|
171
|
+
// conjunctive fixture (the Leonardo one) finds them at ranks 0 and 1. Scope:
|
|
172
|
+
// these are the repo's conjunctive fixtures, and no more.
|
|
149
173
|
const streams: Stream[] = [];
|
|
150
174
|
const rankedCapped = ranked.length > pre.k ? ranked.slice(0, pre.k) : ranked;
|
|
151
175
|
// CONJUNCTIVITY EARLY-EXIT: a conjunctive query's top-ranked anchors
|
|
@@ -98,6 +98,7 @@ export async function resolveConnectors(
|
|
|
98
98
|
});
|
|
99
99
|
const bridgePair = async (l: number, r: number) => {
|
|
100
100
|
if (l === r || links.has(l + "," + r)) return;
|
|
101
|
+
if (ctx.meter) ctx.meter.coverBridges++;
|
|
101
102
|
const link = await bridge(ctx, read(ctx, l), read(ctx, r));
|
|
102
103
|
if (link !== null) links.set(l + "," + r, link);
|
|
103
104
|
};
|
|
@@ -134,6 +135,10 @@ export async function resolveConnectors(
|
|
|
134
135
|
// plus one W-quantum of glue per joint — pass that allowance so the
|
|
135
136
|
// bridge's phrase-scale cap admits the whole learnt run.
|
|
136
137
|
const allowance = middleBytes + (m + 1) * W;
|
|
138
|
+
if (ctx.meter) {
|
|
139
|
+
ctx.meter.coverBridges++;
|
|
140
|
+
ctx.meter.coverAllowanceBytes += allowance;
|
|
141
|
+
}
|
|
137
142
|
const interior = await bridge(
|
|
138
143
|
ctx,
|
|
139
144
|
first.bytes,
|
|
@@ -276,9 +276,36 @@ export async function recallByResonance(
|
|
|
276
276
|
// consensus", while breadth is the SCALE-INVARIANT reading — "a point whose
|
|
277
277
|
// breadth clears `dominates` (> half the query's regions corroborate it) is
|
|
278
278
|
// real consensus; one that does not is a coincidental single-region echo".
|
|
279
|
-
//
|
|
280
|
-
// comparing a POOLED SUM against a floor that prices ONE region's
|
|
281
|
-
// is a dimensional error.
|
|
279
|
+
// THIS USED TO CLAIM A DIMENSIONAL ERROR, AND THAT CLAIM WAS FALSE.
|
|
280
|
+
// It read: "comparing a POOLED SUM against a floor that prices ONE region's
|
|
281
|
+
// evidence is a dimensional error." `consensusFloor` is not priced for one
|
|
282
|
+
// region: thresholds.md §2 derives it as the POOLED-vote significance floor —
|
|
283
|
+
// "each region contributes at most ln(N/c) <= ln(N); ln(N)+1/2 demands ..." —
|
|
284
|
+
// and attention.ts says the same where it builds the vote ("the scale
|
|
285
|
+
// consensusFloor is derived for"). The comparison is in ONE dimension, and
|
|
286
|
+
// it is so because the climb WEIGHTS BY IDF: `wf` in voteRegions is
|
|
287
|
+
// `direct ? df : combined ? idf + df : idf`, and the engine only ever runs the
|
|
288
|
+
// last one (DFMode's default "inverse", the mode every non-test caller uses —
|
|
289
|
+
// `direct` and `combined` are exercised by test/24 and test/27 only, and
|
|
290
|
+
// test/24 pins that their votes DO differ). In those two the sum would leave
|
|
291
|
+
// the floor's dimension and the floor would need re-deriving.
|
|
292
|
+
//
|
|
293
|
+
// What the OR below is really for is SCALE, not dimension (the paragraph
|
|
294
|
+
// above says it): a vote that clears ln(N)+1/2 means "strong" on a small store
|
|
295
|
+
// and "weak" on a large one for the same genuine consensus, so the
|
|
296
|
+
// scale-invariant breadth reading is added beside it.
|
|
297
|
+
//
|
|
298
|
+
// AND THE PREMISE IS IDF. The deviation in the other two weighting modes is
|
|
299
|
+
// TWO-SIDED and DERIVED: `direct` DEFLATES a region (ln(1+c) < ln(N/c) for
|
|
300
|
+
// small c) and `combined` INFLATES it (ln N + ln(1+1/c)), both by at most
|
|
301
|
+
// `ln 2` — see `geometry.ts`'s `consensusFloor`, where the bound lives.
|
|
302
|
+
// MEASURED on 8 anchors across 5 queries, running the same climb in all
|
|
303
|
+
// three modes: ZERO gate inversions — every anchor's `vote >= floor` verdict
|
|
304
|
+
// is the same in `inverse`, `direct` and `combined`, even where the readings
|
|
305
|
+
// straddle the floor on opposite sides (#148: inverse 3.39, combined 4.71
|
|
306
|
+
// above it, direct 1.31 below). Pinned by test/55's test 20. The bar is not
|
|
307
|
+
// re-derived for those modes because nothing reachable needs it; the premise
|
|
308
|
+
// is IDF, and that is now written where the gate reads it.
|
|
282
309
|
//
|
|
283
310
|
// Measured on the 15.7M-node store (N=325,615, so the old floor was 13.19).
|
|
284
311
|
// The absolute vote cannot separate right from wrong at this scale, and the
|
|
@@ -348,7 +375,7 @@ export async function recallByResonance(
|
|
|
348
375
|
if (
|
|
349
376
|
forest.length > 0 &&
|
|
350
377
|
!allWindowsAreScaffolding(ctx, query) &&
|
|
351
|
-
(forest[0].
|
|
378
|
+
(forest[0].idfVote >= minVote || // the IDF sum: the bar's own quantity
|
|
352
379
|
(dominates(forest[0].breadth, 1) && forest[0].peak > Math.LN2))
|
|
353
380
|
) {
|
|
354
381
|
const g = await project(ctx, forest[0].anchor, queryGist);
|
|
@@ -602,6 +629,7 @@ export const recallMechanism: PipelineMechanism = {
|
|
|
602
629
|
moves: r.moves,
|
|
603
630
|
unexplained: r.unexplained,
|
|
604
631
|
provenance: r.echoed ? "recall-echo" : "recall",
|
|
632
|
+
used: new Set<number>(),
|
|
605
633
|
...(r.complete ? { complete: true } : {}),
|
|
606
634
|
}];
|
|
607
635
|
},
|
package/src/mind/mind.ts
CHANGED
|
@@ -11,6 +11,8 @@
|
|
|
11
11
|
|
|
12
12
|
import { cosine, makeKeyring, rng, setVecConfig, Vec } from "../vec.js";
|
|
13
13
|
import { bindSeat, fold, Sema, Space } from "../sema.js";
|
|
14
|
+
import { sampleCorpus, searchCorpus } from "./corpus.js";
|
|
15
|
+
import type { CorpusPair, CorpusResult } from "./corpus.js";
|
|
14
16
|
import { Alphabet } from "../alphabet.js";
|
|
15
17
|
import {
|
|
16
18
|
bytesToTree,
|
|
@@ -21,6 +23,7 @@ import {
|
|
|
21
23
|
reachThreshold,
|
|
22
24
|
stackGrids,
|
|
23
25
|
} from "../geometry.js";
|
|
26
|
+
import { keyEnds } from "./canonical.js";
|
|
24
27
|
import type { ContentFold } from "../geometry.js";
|
|
25
28
|
import { BoundedMap, type Store } from "../store.js";
|
|
26
29
|
import { SQliteStore } from "../store-sqlite.js";
|
|
@@ -146,6 +149,7 @@ interface ConversationData {
|
|
|
146
149
|
import type { AttentionRead, MindContext, Recognition } from "./types.js";
|
|
147
150
|
import { changedNodes, liftAnswer, spliceAll } from "./types.js";
|
|
148
151
|
import {
|
|
152
|
+
canonResolve as canonResolveImpl,
|
|
149
153
|
foldTree,
|
|
150
154
|
gistOf,
|
|
151
155
|
inputBytes,
|
|
@@ -194,10 +198,63 @@ import { type CostReport, Meter } from "../meter.js";
|
|
|
194
198
|
|
|
195
199
|
// ── MindOptions ───────────────────────────────────────────────────────────
|
|
196
200
|
|
|
201
|
+
/** A stored pair as TEXT — the text helper's view of {@link CorpusPair}. */
|
|
202
|
+
export interface CorpusTextPair {
|
|
203
|
+
context: string;
|
|
204
|
+
continuation: string;
|
|
205
|
+
contextId: number;
|
|
206
|
+
continuationId: number;
|
|
207
|
+
matchedBytes: number;
|
|
208
|
+
contextTruncated: boolean;
|
|
209
|
+
continuationTruncated: boolean;
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
/** {@link CorpusResult} as text, plus the prose for why nothing matched. The
|
|
213
|
+
* byte layer reports a STATE; saying it in words belongs to the text layer. */
|
|
214
|
+
export interface CorpusTextResult {
|
|
215
|
+
query: string;
|
|
216
|
+
pairs: CorpusTextPair[];
|
|
217
|
+
resolved: number;
|
|
218
|
+
reached: number;
|
|
219
|
+
totalContexts: number;
|
|
220
|
+
browsed: boolean;
|
|
221
|
+
note?: string;
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
/** What the text helper says when the byte layer reports a miss. */
|
|
225
|
+
const CORPUS_NOTE: Record<string, string> = {
|
|
226
|
+
"nothing-resolved":
|
|
227
|
+
"No trained note sits above the parts of that text the mind recognised. " +
|
|
228
|
+
"It addresses content exactly, so try wording closer to something it was " +
|
|
229
|
+
"actually given — or browse the examples instead.",
|
|
230
|
+
"no-continuations":
|
|
231
|
+
"That text reaches stored nodes, but none of them carries a learnt " +
|
|
232
|
+
"continuation.",
|
|
233
|
+
};
|
|
234
|
+
|
|
235
|
+
/** UTF-8 of bytes for display: reuse {@link decodeText} (the mind's own text
|
|
236
|
+
* conversion), then drop the replacement character a byte-boundary cut leaves
|
|
237
|
+
* behind. Much of a real corpus is non-Latin, so that trailing U+FFFD is the
|
|
238
|
+
* common case, not an exotic one — and it is the ONLY thing added here. */
|
|
239
|
+
function previewCorpusText(bytes: Uint8Array): string {
|
|
240
|
+
return decodeText(bytes).replace(/\uFFFD+$/, "").replace(/\s+/g, " ").trim();
|
|
241
|
+
}
|
|
242
|
+
|
|
197
243
|
export interface MindOptions {
|
|
198
244
|
seed?: number;
|
|
199
245
|
recallQueryK?: number;
|
|
200
246
|
haloQueryK?: number;
|
|
247
|
+
/** Branch nodes the pivot sweep may probe — see {@link MindConfig}. */
|
|
248
|
+
pivotProbeK?: number;
|
|
249
|
+
/** Items one rationale step may itemise — see {@link MindConfig}. */
|
|
250
|
+
rationaleSampleK?: number;
|
|
251
|
+
/** Corpus-reading capacities and budgets — see {@link MindConfig}. */
|
|
252
|
+
corpusLimitMax?: number;
|
|
253
|
+
corpusClimbs?: number;
|
|
254
|
+
corpusContextsPerClimb?: number;
|
|
255
|
+
corpusSampleProbes?: number;
|
|
256
|
+
corpusPreviewBytes?: number;
|
|
257
|
+
corpusSampleFloorBytes?: number;
|
|
201
258
|
normalizeEpsilon?: number;
|
|
202
259
|
cosineEpsilon?: number;
|
|
203
260
|
geometry?: Partial<import("../config.js").GeometryConfig>;
|
|
@@ -347,6 +404,20 @@ export class Mind implements MindContext {
|
|
|
347
404
|
* `traverse.ts`'s ONE definition (edge or halo, with its response-scoped
|
|
348
405
|
* cache). The search holds a bare Store and cannot reach that cache itself,
|
|
349
406
|
* so it asks through this hook; a bare host keeps its raw-store fallback. */
|
|
407
|
+
/** The canonical identity for the search (see GraphSearchHost). */
|
|
408
|
+
canonResolve(bytes: Uint8Array): number | null {
|
|
409
|
+
return canonResolveImpl(this, bytes);
|
|
410
|
+
}
|
|
411
|
+
|
|
412
|
+
/** Feed a search refusal into the rationale (see GraphSearchHost). */
|
|
413
|
+
reportSearch(
|
|
414
|
+
name: string,
|
|
415
|
+
parts: ReadonlyArray<Uint8Array>,
|
|
416
|
+
note: string,
|
|
417
|
+
): void {
|
|
418
|
+
this.trace?.step(name, parts.map((b) => rItem(b)), [], note);
|
|
419
|
+
}
|
|
420
|
+
|
|
350
421
|
leadsSomewhere(id: number): boolean {
|
|
351
422
|
return leadsSomewhere(this, id);
|
|
352
423
|
}
|
|
@@ -374,6 +445,11 @@ export class Mind implements MindContext {
|
|
|
374
445
|
* with the most distributional evidence (highest `prevOf` count — the
|
|
375
446
|
* structural manifestation of its halo). When evidence is equal the
|
|
376
447
|
* first-inserted edge wins. */
|
|
448
|
+
/** See {@link GraphSearchHost.contentKeyEnds}. */
|
|
449
|
+
contentKeyEnds(prefix: Uint8Array, tail: Uint8Array): readonly number[] {
|
|
450
|
+
return keyEnds(this, prefix, tail);
|
|
451
|
+
}
|
|
452
|
+
|
|
377
453
|
chooseNext(node: number): number | undefined {
|
|
378
454
|
return chooseNext(this, node, this._edgeGuide);
|
|
379
455
|
}
|
|
@@ -737,6 +813,55 @@ export class Mind implements MindContext {
|
|
|
737
813
|
return decodeText(r.bytes);
|
|
738
814
|
}
|
|
739
815
|
|
|
816
|
+
// ── Reading the trained memory back ─────────────────────────────────────
|
|
817
|
+
|
|
818
|
+
/** Which stored notes does this query REACH? BYTES in, BYTES out — this
|
|
819
|
+
* method has no notion of text or encoding; the text case is
|
|
820
|
+
* {@link searchCorpusText}, which is one caller of this.
|
|
821
|
+
*
|
|
822
|
+
* Exact content addressing through the machinery an answer already uses
|
|
823
|
+
* (see src/mind/corpus.ts): the query's recognised sites are the resolved
|
|
824
|
+
* subtrees, the climb goes up from the biggest, and a result is a context
|
|
825
|
+
* that carries a learnt continuation. Nothing is written and nothing is
|
|
826
|
+
* indexed. */
|
|
827
|
+
searchCorpus(queryBytes: Uint8Array, limit?: number): CorpusResult {
|
|
828
|
+
return searchCorpus(this, queryBytes, limit);
|
|
829
|
+
}
|
|
830
|
+
|
|
831
|
+
/** Browse real pairs. Deterministic: `from` is the caller's own offset in
|
|
832
|
+
* [0,1), so browsing twice with different offsets shows different notes
|
|
833
|
+
* without a random draw. */
|
|
834
|
+
sampleCorpus(limit?: number, from?: number): CorpusResult {
|
|
835
|
+
return sampleCorpus(this, limit, from);
|
|
836
|
+
}
|
|
837
|
+
|
|
838
|
+
/** The TEXT case of {@link searchCorpus}: encode, search, decode. The search
|
|
839
|
+
* itself exists once, in the byte layer above; only the rendering lives
|
|
840
|
+
* here, with the rest of this class's text modality. */
|
|
841
|
+
searchCorpusText(query: string, limit?: number): CorpusTextResult {
|
|
842
|
+
const result = this.searchCorpus(
|
|
843
|
+
new TextEncoder().encode(query),
|
|
844
|
+
limit,
|
|
845
|
+
);
|
|
846
|
+
return {
|
|
847
|
+
query,
|
|
848
|
+
pairs: result.pairs.map((p: CorpusPair): CorpusTextPair => ({
|
|
849
|
+
context: previewCorpusText(p.context),
|
|
850
|
+
continuation: previewCorpusText(p.continuation),
|
|
851
|
+
contextId: p.contextId,
|
|
852
|
+
continuationId: p.continuationId,
|
|
853
|
+
matchedBytes: p.matchedBytes,
|
|
854
|
+
contextTruncated: p.contextTruncated,
|
|
855
|
+
continuationTruncated: p.continuationTruncated,
|
|
856
|
+
})),
|
|
857
|
+
resolved: result.resolved,
|
|
858
|
+
reached: result.reached,
|
|
859
|
+
totalContexts: result.totalContexts,
|
|
860
|
+
browsed: result.browsed,
|
|
861
|
+
note: result.miss === "matched" ? undefined : CORPUS_NOTE[result.miss],
|
|
862
|
+
};
|
|
863
|
+
}
|
|
864
|
+
|
|
740
865
|
// ── Conversation API ────────────────────────────────────────────────────
|
|
741
866
|
|
|
742
867
|
/** Begin a new conversation, optionally restoring from a previously-saved
|
|
@@ -675,6 +675,13 @@ export interface MechanismResult {
|
|
|
675
675
|
bytes: Uint8Array;
|
|
676
676
|
accounted: Array<[number, number]>;
|
|
677
677
|
moves: number;
|
|
678
|
+
/** WHAT THIS ANSWER SPEAKS FOR — the anchors it voices, and therefore the
|
|
679
|
+
* content the reasoner must not pivot back through. Declared by the
|
|
680
|
+
* mechanism about its OWN result, exactly like `accounted`/`unexplained`/
|
|
681
|
+
* `complete`: post-grounding honours the property and NEVER ASKS WHICH
|
|
682
|
+
* MECHANISM SET IT, so the market stays uniform. An EMPTY set is a real
|
|
683
|
+
* declaration — "this answer voices nothing" (recall) — and withholds
|
|
684
|
+
* nothing; omit the field and the pipeline re-recognises the answer. */
|
|
678
685
|
used?: ReadonlySet<number>;
|
|
679
686
|
unexplained: string;
|
|
680
687
|
/** Explicit weight override. When absent, weight = moves + PASS·unaccounted. */
|