@hviana/sema 0.8.1 → 0.8.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/src/config.d.ts +17 -0
- package/dist/src/config.js +18 -0
- package/dist/src/meter.d.ts +25 -0
- package/dist/src/meter.js +44 -0
- package/dist/src/mind/corpus.d.ts +40 -0
- package/dist/src/mind/corpus.js +149 -0
- package/dist/src/mind/graph-search.d.ts +7 -0
- package/dist/src/mind/graph-search.js +235 -24
- package/dist/src/mind/index.d.ts +3 -1
- package/dist/src/mind/index.js +1 -0
- package/dist/src/mind/match.d.ts +8 -3
- package/dist/src/mind/match.js +142 -58
- package/dist/src/mind/mechanisms/cast.js +18 -2
- package/dist/src/mind/mechanisms/cover.js +6 -0
- package/dist/src/mind/mind.d.ts +55 -0
- package/dist/src/mind/mind.js +72 -2
- package/dist/src/mind/pipeline.js +25 -6
- package/dist/src/mind/reasoning.d.ts +5 -1
- package/dist/src/mind/reasoning.js +54 -1
- package/dist/src/mind/traverse.js +9 -1
- package/dist/src/mind/types.d.ts +22 -0
- package/docs/failures/tempting-but-wrong.md +31 -2
- package/jsr.json +1 -1
- package/package.json +1 -1
- package/src/config.ts +35 -0
- package/src/meter.ts +47 -0
- package/src/mind/corpus.ts +202 -0
- package/src/mind/graph-search.ts +252 -23
- package/src/mind/index.ts +8 -1
- package/src/mind/match.ts +143 -54
- package/src/mind/mechanisms/cast.ts +17 -1
- package/src/mind/mechanisms/cover.ts +5 -0
- package/src/mind/mind.ts +123 -0
- package/src/mind/pipeline.ts +30 -6
- package/src/mind/reasoning.ts +55 -0
- package/src/mind/traverse.ts +9 -1
- package/src/mind/types.ts +26 -0
- package/test/100-complete-grounding-trace.test.mjs +109 -0
- package/test/101-alignment-gap-bound.test.mjs +106 -0
- package/test/102-production-composes-at-scale.test.mjs +110 -0
- package/test/103-alignment-gap-budget.test.mjs +89 -0
- package/test/104-composition-is-reported.test.mjs +90 -0
- package/test/105-derive-through-reports-its-refusal.test.mjs +113 -0
- package/test/106-the-join-fires.test.mjs +94 -0
- package/test/107-the-join-is-counted.test.mjs +81 -0
- package/test/108-the-join-chains.test.mjs +78 -0
- package/test/109-the-pivot-is-counted.test.mjs +60 -0
- package/test/110-the-reasoner-stops-when-the-question-is-answered.test.mjs +91 -0
- package/test/111-the-cover-assembly-is-counted.test.mjs +74 -0
- package/test/112-the-exploration-does-not-grow-with-the-hub.test.mjs +89 -0
- package/test/113-the-rationale-payload-is-bounded.test.mjs +84 -0
- package/test/114-alignment-budget-is-per-sweep.test.mjs +93 -0
- package/test/116-the-extension-is-gated-by-the-pipelines-own-remainder.test.mjs +100 -0
- package/test/117-corpus-search.test.mjs +171 -0
- package/test/14-scaling.test.mjs +10 -7
- package/test/76-reference-binding.test.mjs +6 -1
- package/test/89-completion-recursion.test.mjs +30 -5
package/src/mind/match.ts
CHANGED
|
@@ -38,7 +38,7 @@ import {
|
|
|
38
38
|
identityBar,
|
|
39
39
|
significanceBar,
|
|
40
40
|
} from "../geometry.js";
|
|
41
|
-
import { bytesEqual, indexOf } from "../bytes.js";
|
|
41
|
+
import { bytesEqual, indexOf, latin1 } from "../bytes.js";
|
|
42
42
|
import type { MindContext } from "./types.js";
|
|
43
43
|
import { chainReach, leafIdRun } from "./canonical.js";
|
|
44
44
|
import { foldTree, gistOf, perceive, read, resolve } from "./primitives.js";
|
|
@@ -54,6 +54,7 @@ import {
|
|
|
54
54
|
sharedReachMemo,
|
|
55
55
|
} from "./traverse.js";
|
|
56
56
|
import { recognise, segment } from "./recognition.js";
|
|
57
|
+
import { rItem } from "./trace.js";
|
|
57
58
|
import type { Site } from "./graph-search.js";
|
|
58
59
|
|
|
59
60
|
// ═══════════════════════════════════════════════════════════════════════════
|
|
@@ -360,9 +361,14 @@ export interface AlignGap {
|
|
|
360
361
|
|
|
361
362
|
/** Extend a seed match (query offset qo ↔ candidate offset co) to its maximal
|
|
362
363
|
* common run, then walk outward in both directions collecting further common
|
|
363
|
-
* runs of at least W bytes across
|
|
364
|
-
*
|
|
365
|
-
*
|
|
364
|
+
* runs of at least W bytes across mismatch gaps. Each gap's LENGTH is the
|
|
365
|
+
* pair's own extent (a gap cannot be longer than the bytes it spans) and the
|
|
366
|
+
* sweep's WORK is proportional to the bytes a run spans (the context's windows
|
|
367
|
+
* are indexed once, then the query's are walked) — the arity bound
|
|
368
|
+
* (`chainReach`) used to cap BOTH, and truncated every learned frame whose
|
|
369
|
+
* slot was longer. Each sweep owns its own budget, so an exhausted right
|
|
370
|
+
* sweep never starves the left one. Returns the matched query spans and the
|
|
371
|
+
* mismatch pairs between consecutive runs.
|
|
366
372
|
*
|
|
367
373
|
* This is the SEEDED aligner, distinct from {@link alignRuns}: that one finds
|
|
368
374
|
* every run two structures share anywhere (a weave), this one reads two
|
|
@@ -382,7 +388,21 @@ export function alignAround(
|
|
|
382
388
|
co: number,
|
|
383
389
|
): { matched: Array<[number, number]>; gaps: AlignGap[] } {
|
|
384
390
|
const W = ctx.space.maxGroup;
|
|
385
|
-
|
|
391
|
+
// THE GAP LENGTH IS THE PAIR'S OWN EXTENT; THE WORK IS BUDGETED.
|
|
392
|
+
//
|
|
393
|
+
// The sweep walks (queryGap, contextGap) pairs by ASCENDING total, so reaching
|
|
394
|
+
// a gap of size G costs about G²/2 pairs. Bounding the LENGTH by the write
|
|
395
|
+
// side's arity (`chainReach(W)` = 16) therefore truncated every learned frame
|
|
396
|
+
// whose slot is longer — measured: `bindReference` reported the cap at 18, 24,
|
|
397
|
+
// 30 and 36 bytes and `recall` answered with ANOTHER instance's filler — while
|
|
398
|
+
// removing the bound outright took the corpus-cost guard (test/89) from
|
|
399
|
+
// milliseconds to 68 seconds.
|
|
400
|
+
//
|
|
401
|
+
// Bounding the PAIRS keeps a call's cost constant however long the pair is,
|
|
402
|
+
// and the ascending order means an exhausted budget drops the FAR
|
|
403
|
+
// continuations and never the near ones — the same degradation recognition.ts
|
|
404
|
+
// documents for its canon budget. Length and work are different questions;
|
|
405
|
+
// this is the one place they were conflated.
|
|
386
406
|
// Maximal run around the seed.
|
|
387
407
|
let qs = qo, ss = co;
|
|
388
408
|
while (qs > 0 && ss > 0 && q[qs - 1] === c[ss - 1]) {
|
|
@@ -396,8 +416,60 @@ export function alignAround(
|
|
|
396
416
|
}
|
|
397
417
|
const matched: Array<[number, number]> = [[qs, qe]];
|
|
398
418
|
const gaps: AlignGap[] = [];
|
|
399
|
-
//
|
|
400
|
-
//
|
|
419
|
+
// THE SWEEP IS STRUCTURAL, NOT ENUMERATIVE.
|
|
420
|
+
//
|
|
421
|
+
// The criterion is unchanged: the next common run, MINIMUM TOTAL GAP, ties to
|
|
422
|
+
// the smaller query gap. What changed is how it is found. Enumerating
|
|
423
|
+
// (queryGap, contextGap) pairs by ascending total reaches a run at total t in
|
|
424
|
+
// about t²/2 pairs — and that quadratic shape, not the reach, was the cost
|
|
425
|
+
// problem: capping the pairs dropped reach (a legitimate 24-byte slot stopped
|
|
426
|
+
// being found), while leaving them uncapped cost 68 seconds on the corpus
|
|
427
|
+
// guard. Neither is the answer, because the answer is the algorithm.
|
|
428
|
+
//
|
|
429
|
+
// The context's windows are indexed ONCE, for lengths 1..W — W being the
|
|
430
|
+
// geometry's own unit of composition, so nothing is chosen here. Each step
|
|
431
|
+
// then walks the query's windows outward from the anchor: for a given query
|
|
432
|
+
// gap the nearest context gap that continues a run is one O(1) lookup, and the
|
|
433
|
+
// walk stops the moment the query gap alone exceeds the best total already
|
|
434
|
+
// found. So the work is proportional to the bytes the run SPANS. No budget,
|
|
435
|
+
// no cap, no number: a long slot is reached, and its price is already the
|
|
436
|
+
// ladder's (its bytes are unaccounted, so the search pays PASS per byte).
|
|
437
|
+
const index: Array<Map<string, number[]>> = [];
|
|
438
|
+
for (let len = 1; len <= W; len++) {
|
|
439
|
+
const m = new Map<string, number[]>();
|
|
440
|
+
for (let o = 0; o + len <= c.length; o++) {
|
|
441
|
+
const key = latin1(c.subarray(o, o + len));
|
|
442
|
+
const at = m.get(key);
|
|
443
|
+
if (at === undefined) m.set(key, [o]);
|
|
444
|
+
else at.push(o);
|
|
445
|
+
}
|
|
446
|
+
index.push(m);
|
|
447
|
+
}
|
|
448
|
+
/** Smallest listed offset at or after `from`, or -1. */
|
|
449
|
+
const fromAt = (list: number[], from: number): number => {
|
|
450
|
+
let lo = 0, hi = list.length - 1, best = -1;
|
|
451
|
+
while (lo <= hi) {
|
|
452
|
+
const mid = (lo + hi) >> 1;
|
|
453
|
+
if (list[mid] >= from) {
|
|
454
|
+
best = list[mid];
|
|
455
|
+
hi = mid - 1;
|
|
456
|
+
} else lo = mid + 1;
|
|
457
|
+
}
|
|
458
|
+
return best;
|
|
459
|
+
};
|
|
460
|
+
/** Largest listed offset at or before `to`, or -1. */
|
|
461
|
+
const toAt = (list: number[], to: number): number => {
|
|
462
|
+
let lo = 0, hi = list.length - 1, best = -1;
|
|
463
|
+
while (lo <= hi) {
|
|
464
|
+
const mid = (lo + hi) >> 1;
|
|
465
|
+
if (list[mid] <= to) {
|
|
466
|
+
best = list[mid];
|
|
467
|
+
lo = mid + 1;
|
|
468
|
+
} else hi = mid - 1;
|
|
469
|
+
}
|
|
470
|
+
return best;
|
|
471
|
+
};
|
|
472
|
+
/** Length of the common run STARTING at (qi, si). */
|
|
401
473
|
const runLenAt = (qi: number, si: number): number => {
|
|
402
474
|
let n = 0;
|
|
403
475
|
while (qi + n < q.length && si + n < c.length && q[qi + n] === c[si + n]) {
|
|
@@ -405,63 +477,80 @@ export function alignAround(
|
|
|
405
477
|
}
|
|
406
478
|
return n;
|
|
407
479
|
};
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
|
|
421
|
-
|
|
480
|
+
/** Length of the common run ENDING at (qi, si). */
|
|
481
|
+
const runLenBefore = (qi: number, si: number): number => {
|
|
482
|
+
let n = 0;
|
|
483
|
+
while (n < qi && n < si && q[qi - 1 - n] === c[si - 1 - n]) n++;
|
|
484
|
+
return n;
|
|
485
|
+
};
|
|
486
|
+
/** The next run outward from an anchor, or null when the bytes run out. */
|
|
487
|
+
const nextRun = (
|
|
488
|
+
qi: number,
|
|
489
|
+
si: number,
|
|
490
|
+
forward: boolean,
|
|
491
|
+
): { gq: number; gs: number; n: number } | null => {
|
|
492
|
+
const qLim = forward ? q.length - qi : qi;
|
|
493
|
+
let best: { gq: number; gs: number; n: number } | null = null;
|
|
494
|
+
for (let gq = 0; gq < qLim; gq++) {
|
|
495
|
+
// No later query gap can beat a total already found.
|
|
496
|
+
if (best !== null && gq > best.gq + best.gs) break;
|
|
497
|
+
const left = qLim - gq;
|
|
498
|
+
// A run of >= W bytes, or — when the query itself ends inside one window —
|
|
499
|
+
// the run that REACHES that end. Exactly the acceptance the sweep had.
|
|
500
|
+
const lens = left >= W ? [W] : [left];
|
|
501
|
+
for (const len of lens) {
|
|
502
|
+
const key = latin1(
|
|
503
|
+
q.subarray(
|
|
504
|
+
forward ? qi + gq : qi - gq - len,
|
|
505
|
+
forward ? qi + gq + len : qi - gq,
|
|
506
|
+
),
|
|
507
|
+
);
|
|
508
|
+
const list = index[len - 1].get(key);
|
|
509
|
+
if (list === undefined) continue;
|
|
510
|
+
const o = forward ? fromAt(list, si) : toAt(list, si - len);
|
|
511
|
+
if (o < 0) continue;
|
|
512
|
+
const n = forward
|
|
513
|
+
? runLenAt(qi + gq, o)
|
|
514
|
+
: runLenBefore(qi - gq, o + len);
|
|
515
|
+
if (n < 1) continue;
|
|
516
|
+
if (
|
|
517
|
+
forward ? n >= W || qi + gq + n === q.length : n >= W || n === qi - gq
|
|
518
|
+
) {
|
|
519
|
+
const gs = forward ? o - si : si - len - o;
|
|
520
|
+
if (best === null || gq + gs < best.gq + best.gs) {
|
|
521
|
+
best = { gq, gs, n };
|
|
422
522
|
}
|
|
423
|
-
matched.push([qi + gq, qi + gq + n]);
|
|
424
|
-
qi = qi + gq + n;
|
|
425
|
-
si = si + gs + n;
|
|
426
|
-
found = true;
|
|
427
523
|
break;
|
|
428
524
|
}
|
|
429
525
|
}
|
|
430
526
|
}
|
|
431
|
-
|
|
527
|
+
return best;
|
|
528
|
+
};
|
|
529
|
+
// RIGHT sweep.
|
|
530
|
+
let qi = qe, si = se;
|
|
531
|
+
for (;;) {
|
|
532
|
+
const step = nextRun(qi, si, true);
|
|
533
|
+
if (step === null) break;
|
|
534
|
+
if (step.gq > 0 || step.gs > 0) {
|
|
535
|
+
gaps.push({ qs: qi, qe: qi + step.gq, cs: si, ce: si + step.gs });
|
|
536
|
+
}
|
|
537
|
+
matched.push([qi + step.gq, qi + step.gq + step.n]);
|
|
538
|
+
qi = qi + step.gq + step.n;
|
|
539
|
+
si = si + step.gs + step.n;
|
|
432
540
|
}
|
|
433
|
-
// LEFT sweep (mirror)
|
|
541
|
+
// LEFT sweep (mirror): an independent walk, so an exhausted right side can
|
|
542
|
+
// never starve it (pinned by test/114).
|
|
434
543
|
qi = qs;
|
|
435
544
|
si = ss;
|
|
436
545
|
for (;;) {
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
|
|
441
|
-
if (gs > reachCap) continue;
|
|
442
|
-
if (qi - gq <= 0 || si - gs <= 0) continue;
|
|
443
|
-
// Run ENDING at (qi - gq, si - gs).
|
|
444
|
-
let n = 0;
|
|
445
|
-
while (
|
|
446
|
-
n < qi - gq && n < si - gs &&
|
|
447
|
-
q[qi - gq - 1 - n] === c[si - gs - 1 - n]
|
|
448
|
-
) {
|
|
449
|
-
n++;
|
|
450
|
-
}
|
|
451
|
-
if (n >= W || n === qi - gq) {
|
|
452
|
-
if (n === 0) continue;
|
|
453
|
-
if (gq > 0 || gs > 0) {
|
|
454
|
-
gaps.push({ qs: qi - gq, qe: qi, cs: si - gs, ce: si });
|
|
455
|
-
}
|
|
456
|
-
matched.push([qi - gq - n, qi - gq]);
|
|
457
|
-
qi = qi - gq - n;
|
|
458
|
-
si = si - gs - n;
|
|
459
|
-
found = true;
|
|
460
|
-
break;
|
|
461
|
-
}
|
|
462
|
-
}
|
|
546
|
+
const step = nextRun(qi, si, false);
|
|
547
|
+
if (step === null) break;
|
|
548
|
+
if (step.gq > 0 || step.gs > 0) {
|
|
549
|
+
gaps.push({ qs: qi - step.gq, qe: qi, cs: si - step.gs, ce: si });
|
|
463
550
|
}
|
|
464
|
-
|
|
551
|
+
matched.push([qi - step.gq - step.n, qi - step.gq]);
|
|
552
|
+
qi = qi - step.gq - step.n;
|
|
553
|
+
si = si - step.gs - step.n;
|
|
465
554
|
}
|
|
466
555
|
return { matched, gaps };
|
|
467
556
|
}
|
|
@@ -615,7 +615,23 @@ export async function counterfactualTransfer(
|
|
|
615
615
|
fwd !== null && indexOf(answer, fwd, 0) < 0 &&
|
|
616
616
|
!restatesQuery(query, fwd)
|
|
617
617
|
) {
|
|
618
|
-
|
|
618
|
+
// THROUGH THE SHARED JOINER, not a bare concatenation.
|
|
619
|
+
//
|
|
620
|
+
// `joinWithBridge` is the composition step every out-of-search assembly
|
|
621
|
+
// shares (multi-topic fusion, CAST's substitution and comparison): it
|
|
622
|
+
// asks the corpus for a learnt connector between the pieces and, on a
|
|
623
|
+
// miss, joins them BARE **and says so** — the `bridgeMiss` step (see
|
|
624
|
+
// resonance.ts). This site bypassed it, and that is the whole of the
|
|
625
|
+
// gluing the study measured: `"Steel is hard"` + `"wet"` came back as
|
|
626
|
+
// `"hardwet"`, `"eva director father"` + `"The father of…"` as
|
|
627
|
+
// `"fatherThe"` — compositions no rationale could show, because the one
|
|
628
|
+
// step that made them left no trace.
|
|
629
|
+
//
|
|
630
|
+
// Routing it through the shared joiner is the instrumentation fix that
|
|
631
|
+
// comes first: a bare join stays possible (the house rule is "joined
|
|
632
|
+
// bare, never silent") but it is now VISIBLE, and an attested connector
|
|
633
|
+
// is used when the corpus has one.
|
|
634
|
+
answer = await joinWithBridge(ctx, answer, fwd);
|
|
619
635
|
}
|
|
620
636
|
ctx.trace?.step(
|
|
621
637
|
"projectCounterfactual",
|
|
@@ -98,6 +98,7 @@ export async function resolveConnectors(
|
|
|
98
98
|
});
|
|
99
99
|
const bridgePair = async (l: number, r: number) => {
|
|
100
100
|
if (l === r || links.has(l + "," + r)) return;
|
|
101
|
+
if (ctx.meter) ctx.meter.coverBridges++;
|
|
101
102
|
const link = await bridge(ctx, read(ctx, l), read(ctx, r));
|
|
102
103
|
if (link !== null) links.set(l + "," + r, link);
|
|
103
104
|
};
|
|
@@ -134,6 +135,10 @@ export async function resolveConnectors(
|
|
|
134
135
|
// plus one W-quantum of glue per joint — pass that allowance so the
|
|
135
136
|
// bridge's phrase-scale cap admits the whole learnt run.
|
|
136
137
|
const allowance = middleBytes + (m + 1) * W;
|
|
138
|
+
if (ctx.meter) {
|
|
139
|
+
ctx.meter.coverBridges++;
|
|
140
|
+
ctx.meter.coverAllowanceBytes += allowance;
|
|
141
|
+
}
|
|
137
142
|
const interior = await bridge(
|
|
138
143
|
ctx,
|
|
139
144
|
first.bytes,
|
package/src/mind/mind.ts
CHANGED
|
@@ -11,9 +11,12 @@
|
|
|
11
11
|
|
|
12
12
|
import { cosine, makeKeyring, rng, setVecConfig, Vec } from "../vec.js";
|
|
13
13
|
import { bindSeat, fold, Sema, Space } from "../sema.js";
|
|
14
|
+
import { sampleCorpus, searchCorpus } from "./corpus.js";
|
|
15
|
+
import type { CorpusPair, CorpusResult } from "./corpus.js";
|
|
14
16
|
import { Alphabet } from "../alphabet.js";
|
|
15
17
|
import {
|
|
16
18
|
bytesToTree,
|
|
19
|
+
contentBoundaries,
|
|
17
20
|
contentFoldIncremental,
|
|
18
21
|
Grid,
|
|
19
22
|
gridToTree,
|
|
@@ -146,6 +149,7 @@ interface ConversationData {
|
|
|
146
149
|
import type { AttentionRead, MindContext, Recognition } from "./types.js";
|
|
147
150
|
import { changedNodes, liftAnswer, spliceAll } from "./types.js";
|
|
148
151
|
import {
|
|
152
|
+
canonResolve as canonResolveImpl,
|
|
149
153
|
foldTree,
|
|
150
154
|
gistOf,
|
|
151
155
|
inputBytes,
|
|
@@ -194,10 +198,61 @@ import { type CostReport, Meter } from "../meter.js";
|
|
|
194
198
|
|
|
195
199
|
// ── MindOptions ───────────────────────────────────────────────────────────
|
|
196
200
|
|
|
201
|
+
/** A stored pair as TEXT — the text helper's view of {@link CorpusPair}. */
|
|
202
|
+
export interface CorpusTextPair {
|
|
203
|
+
context: string;
|
|
204
|
+
continuation: string;
|
|
205
|
+
contextId: number;
|
|
206
|
+
continuationId: number;
|
|
207
|
+
matchedBytes: number;
|
|
208
|
+
contextTruncated: boolean;
|
|
209
|
+
continuationTruncated: boolean;
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
/** {@link CorpusResult} as text, plus the prose for why nothing matched. The
|
|
213
|
+
* byte layer reports a STATE; saying it in words belongs to the text layer. */
|
|
214
|
+
export interface CorpusTextResult {
|
|
215
|
+
query: string;
|
|
216
|
+
pairs: CorpusTextPair[];
|
|
217
|
+
resolved: number;
|
|
218
|
+
reached: number;
|
|
219
|
+
totalContexts: number;
|
|
220
|
+
browsed: boolean;
|
|
221
|
+
note?: string;
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
/** What the text helper says when the byte layer reports a miss. */
|
|
225
|
+
const CORPUS_NOTE: Record<string, string> = {
|
|
226
|
+
"nothing-resolved":
|
|
227
|
+
"No trained note sits above the parts of that text the mind recognised. " +
|
|
228
|
+
"It addresses content exactly, so try wording closer to something it was " +
|
|
229
|
+
"actually given — or browse the examples instead.",
|
|
230
|
+
"no-continuations":
|
|
231
|
+
"That text reaches stored nodes, but none of them carries a learnt " +
|
|
232
|
+
"continuation.",
|
|
233
|
+
};
|
|
234
|
+
|
|
235
|
+
/** UTF-8 of bytes for display: reuse {@link decodeText} (the mind's own text
|
|
236
|
+
* conversion), then drop the replacement character a byte-boundary cut leaves
|
|
237
|
+
* behind. Much of a real corpus is non-Latin, so that trailing U+FFFD is the
|
|
238
|
+
* common case, not an exotic one — and it is the ONLY thing added here. */
|
|
239
|
+
function previewCorpusText(bytes: Uint8Array): string {
|
|
240
|
+
return decodeText(bytes).replace(/\uFFFD+$/, "").replace(/\s+/g, " ").trim();
|
|
241
|
+
}
|
|
242
|
+
|
|
197
243
|
export interface MindOptions {
|
|
198
244
|
seed?: number;
|
|
199
245
|
recallQueryK?: number;
|
|
200
246
|
haloQueryK?: number;
|
|
247
|
+
/** Items one rationale step may itemise — see {@link MindConfig}. */
|
|
248
|
+
rationaleSampleK?: number;
|
|
249
|
+
/** Corpus-reading capacities and budgets — see {@link MindConfig}. */
|
|
250
|
+
corpusLimitMax?: number;
|
|
251
|
+
corpusClimbs?: number;
|
|
252
|
+
corpusContextsPerClimb?: number;
|
|
253
|
+
corpusSampleProbes?: number;
|
|
254
|
+
corpusPreviewBytes?: number;
|
|
255
|
+
corpusSampleFloorBytes?: number;
|
|
201
256
|
normalizeEpsilon?: number;
|
|
202
257
|
cosineEpsilon?: number;
|
|
203
258
|
geometry?: Partial<import("../config.js").GeometryConfig>;
|
|
@@ -347,6 +402,20 @@ export class Mind implements MindContext {
|
|
|
347
402
|
* `traverse.ts`'s ONE definition (edge or halo, with its response-scoped
|
|
348
403
|
* cache). The search holds a bare Store and cannot reach that cache itself,
|
|
349
404
|
* so it asks through this hook; a bare host keeps its raw-store fallback. */
|
|
405
|
+
/** The canonical identity for the search (see GraphSearchHost). */
|
|
406
|
+
canonResolve(bytes: Uint8Array): number | null {
|
|
407
|
+
return canonResolveImpl(this, bytes);
|
|
408
|
+
}
|
|
409
|
+
|
|
410
|
+
/** Feed a search refusal into the rationale (see GraphSearchHost). */
|
|
411
|
+
reportSearch(
|
|
412
|
+
name: string,
|
|
413
|
+
parts: ReadonlyArray<Uint8Array>,
|
|
414
|
+
note: string,
|
|
415
|
+
): void {
|
|
416
|
+
this.trace?.step(name, parts.map((b) => rItem(b)), [], note);
|
|
417
|
+
}
|
|
418
|
+
|
|
350
419
|
leadsSomewhere(id: number): boolean {
|
|
351
420
|
return leadsSomewhere(this, id);
|
|
352
421
|
}
|
|
@@ -374,6 +443,11 @@ export class Mind implements MindContext {
|
|
|
374
443
|
* with the most distributional evidence (highest `prevOf` count — the
|
|
375
444
|
* structural manifestation of its halo). When evidence is equal the
|
|
376
445
|
* first-inserted edge wins. */
|
|
446
|
+
/** See {@link GraphSearchHost.contentCuts}. */
|
|
447
|
+
contentCuts(bytes: Uint8Array): readonly number[] {
|
|
448
|
+
return contentBoundaries(this.space, bytes);
|
|
449
|
+
}
|
|
450
|
+
|
|
377
451
|
chooseNext(node: number): number | undefined {
|
|
378
452
|
return chooseNext(this, node, this._edgeGuide);
|
|
379
453
|
}
|
|
@@ -737,6 +811,55 @@ export class Mind implements MindContext {
|
|
|
737
811
|
return decodeText(r.bytes);
|
|
738
812
|
}
|
|
739
813
|
|
|
814
|
+
// ── Reading the trained memory back ─────────────────────────────────────
|
|
815
|
+
|
|
816
|
+
/** Which stored notes does this query REACH? BYTES in, BYTES out — this
|
|
817
|
+
* method has no notion of text or encoding; the text case is
|
|
818
|
+
* {@link searchCorpusText}, which is one caller of this.
|
|
819
|
+
*
|
|
820
|
+
* Exact content addressing through the machinery an answer already uses
|
|
821
|
+
* (see src/mind/corpus.ts): the query's recognised sites are the resolved
|
|
822
|
+
* subtrees, the climb goes up from the biggest, and a result is a context
|
|
823
|
+
* that carries a learnt continuation. Nothing is written and nothing is
|
|
824
|
+
* indexed. */
|
|
825
|
+
searchCorpus(queryBytes: Uint8Array, limit?: number): CorpusResult {
|
|
826
|
+
return searchCorpus(this, queryBytes, limit);
|
|
827
|
+
}
|
|
828
|
+
|
|
829
|
+
/** Browse real pairs. Deterministic: `from` is the caller's own offset in
|
|
830
|
+
* [0,1), so browsing twice with different offsets shows different notes
|
|
831
|
+
* without a random draw. */
|
|
832
|
+
sampleCorpus(limit?: number, from?: number): CorpusResult {
|
|
833
|
+
return sampleCorpus(this, limit, from);
|
|
834
|
+
}
|
|
835
|
+
|
|
836
|
+
/** The TEXT case of {@link searchCorpus}: encode, search, decode. The search
|
|
837
|
+
* itself exists once, in the byte layer above; only the rendering lives
|
|
838
|
+
* here, with the rest of this class's text modality. */
|
|
839
|
+
searchCorpusText(query: string, limit?: number): CorpusTextResult {
|
|
840
|
+
const result = this.searchCorpus(
|
|
841
|
+
new TextEncoder().encode(query),
|
|
842
|
+
limit,
|
|
843
|
+
);
|
|
844
|
+
return {
|
|
845
|
+
query,
|
|
846
|
+
pairs: result.pairs.map((p: CorpusPair): CorpusTextPair => ({
|
|
847
|
+
context: previewCorpusText(p.context),
|
|
848
|
+
continuation: previewCorpusText(p.continuation),
|
|
849
|
+
contextId: p.contextId,
|
|
850
|
+
continuationId: p.continuationId,
|
|
851
|
+
matchedBytes: p.matchedBytes,
|
|
852
|
+
contextTruncated: p.contextTruncated,
|
|
853
|
+
continuationTruncated: p.continuationTruncated,
|
|
854
|
+
})),
|
|
855
|
+
resolved: result.resolved,
|
|
856
|
+
reached: result.reached,
|
|
857
|
+
totalContexts: result.totalContexts,
|
|
858
|
+
browsed: result.browsed,
|
|
859
|
+
note: result.miss === "matched" ? undefined : CORPUS_NOTE[result.miss],
|
|
860
|
+
};
|
|
861
|
+
}
|
|
862
|
+
|
|
740
863
|
// ── Conversation API ────────────────────────────────────────────────────
|
|
741
864
|
|
|
742
865
|
/** Begin a new conversation, optionally restoring from a previously-saved
|
package/src/mind/pipeline.ts
CHANGED
|
@@ -545,12 +545,40 @@ export async function think(
|
|
|
545
545
|
ctx.store.nextFirst(id, hubBound(ctx)).map((n) => read(ctx, n))
|
|
546
546
|
)
|
|
547
547
|
: [];
|
|
548
|
+
// REPORTABLE, NOT SILENT. A declared-complete grounding ends the derivation
|
|
549
|
+
// here, and that decision is part of the derivation's shape: the reader of a
|
|
550
|
+
// rationale must be able to see that the chain stopped because the mechanism
|
|
551
|
+
// claimed the query WAS the context, not because nothing followed. The step
|
|
552
|
+
// carries the claim, not a re-description of the answer — the extension is
|
|
553
|
+
// skipped, so there is no output item to show.
|
|
554
|
+
if (decided.complete) {
|
|
555
|
+
ctx.trace?.step(
|
|
556
|
+
"completeGrounding",
|
|
557
|
+
[rItem(answer, provenance)],
|
|
558
|
+
[],
|
|
559
|
+
"grounding declared complete — the query IS the context, so " +
|
|
560
|
+
"post-grounding extension is skipped",
|
|
561
|
+
);
|
|
562
|
+
}
|
|
563
|
+
// THE REASONER JUDGES ITS OWN EXTENSIONS BY THE PIPELINE'S REMAINDER, not by
|
|
564
|
+
// the ladder's `accounted` — and by the SAME reading the fuse gate below uses,
|
|
565
|
+
// with the same W floor. `accounted` is a COST quantity (measured: a query
|
|
566
|
+
// fully explained by one computed span plus bridged connectors reports
|
|
567
|
+
// `accounted: []` while nothing is unexplained), and a remainder under one
|
|
568
|
+
// river-fold quantum is bridging punctuation, never a second topic — so it
|
|
569
|
+
// licenses no extension and blocks none.
|
|
570
|
+
const explained: Array<[number, number]> = [
|
|
571
|
+
...decided.accounted,
|
|
572
|
+
...pre.computed.map((u): [number, number] => [u.i, u.j]),
|
|
573
|
+
];
|
|
574
|
+
const uncovered = unexplainedSpans(query.length, explained)
|
|
575
|
+
.filter(([a, b]) => b - a >= ctx.space.maxGroup);
|
|
548
576
|
const reasoned = decided.complete ? answer : meter
|
|
549
577
|
? await meter.time(
|
|
550
578
|
"reason",
|
|
551
|
-
() => reason(ctx, query, answer, preConsumed, pre, voiced),
|
|
579
|
+
() => reason(ctx, query, answer, preConsumed, pre, voiced, uncovered),
|
|
552
580
|
)
|
|
553
|
-
: await reason(ctx, query, answer, preConsumed, pre, voiced);
|
|
581
|
+
: await reason(ctx, query, answer, preConsumed, pre, voiced, uncovered);
|
|
554
582
|
|
|
555
583
|
// Fuse only when the query has a genuine REMAINDER no mechanism's
|
|
556
584
|
// structural evidence touched at all. `decided.accounted` alone
|
|
@@ -568,10 +596,6 @@ export async function think(
|
|
|
568
596
|
// observed: a single space between two fully-computed arithmetic spans
|
|
569
597
|
// ("2+2 3+3") registered as "unaccounted" and pulled in an unrelated
|
|
570
598
|
// corpus fact, corrupting "4 6" into "4 63".
|
|
571
|
-
const explained: Array<[number, number]> = [
|
|
572
|
-
...decided.accounted,
|
|
573
|
-
...pre.computed.map((u): [number, number] => [u.i, u.j]),
|
|
574
|
-
];
|
|
575
599
|
const remainder = unaccounted(explained);
|
|
576
600
|
// Whether the winning candidate's entire recognised substance is
|
|
577
601
|
// COMPUTED — every accounted span exactly a pre.computed span, nothing
|
package/src/mind/reasoning.ts
CHANGED
|
@@ -43,6 +43,10 @@ export async function reason(
|
|
|
43
43
|
preConsumed: ReadonlySet<number>,
|
|
44
44
|
pre: Precomputed,
|
|
45
45
|
voiced: readonly Uint8Array[] = [],
|
|
46
|
+
/** The query material the GROUNDING left uncovered — the cost ladder's own
|
|
47
|
+
* `unaccounted` spans. Only the reasoner's OWN extensions are judged
|
|
48
|
+
* against it; a mechanism carrying its own `used` set owns its shape. */
|
|
49
|
+
uncovered: readonly (readonly [number, number])[] = [],
|
|
46
50
|
): Promise<Uint8Array> {
|
|
47
51
|
// Echo guard: a query that is ITSELF a learnt continuation (some context's
|
|
48
52
|
// answer) is being asked back at the system — hopping forward from it would
|
|
@@ -200,6 +204,57 @@ export async function reason(
|
|
|
200
204
|
const fc = await follow(ctx, pivot, qv);
|
|
201
205
|
consumeAll(pivot);
|
|
202
206
|
if (fc === null || bytesEqual(fc, cur) || restatesQuery(query, fc)) break;
|
|
207
|
+
// WHOSE EXTENSION IS THIS?
|
|
208
|
+
//
|
|
209
|
+
// `voiced` is what the mechanism WITHHELD (the pipeline sends the used
|
|
210
|
+
// anchors' CONTINUATIONS, not their bytes — see pipeline's own note), so a
|
|
211
|
+
// non-empty `voiced` means exactly what that note says: the grounding came
|
|
212
|
+
// from a mechanism that carries its own short `used` set (cast/join) and
|
|
213
|
+
// therefore owns the shape of its answer. The further terms inside such a
|
|
214
|
+
// seat are legitimately followable — test/29 C3's `Mona Lisa` lives inside
|
|
215
|
+
// the voiced seat and leads on to a fact about neither analog.
|
|
216
|
+
//
|
|
217
|
+
// Every other grounding is ordinary, and an extension of it is the
|
|
218
|
+
// reasoner's own inference: it is taken only while question material the
|
|
219
|
+
// grounding left uncovered remains AND the step carries some of it, judged
|
|
220
|
+
// by the mind's own line between chance and evidence — one W-byte window,
|
|
221
|
+
// no word notion, no character class, no threshold. Measured: the drift's
|
|
222
|
+
// second step (`the Eiffel Tower is in Paris` after `Paris is famous for
|
|
223
|
+
// the Eiffel Tower`) carries no window of `" famous for"` and is refused,
|
|
224
|
+
// while the first carries it. Terminates by a real argument: the uncovered
|
|
225
|
+
// material is finite and each taken extension must carry some of it.
|
|
226
|
+
const producerOwnsShape = voiced.length > 0;
|
|
227
|
+
if (!producerOwnsShape && uncovered.length > 0) {
|
|
228
|
+
const W = ctx.space.maxGroup;
|
|
229
|
+
let progress = false;
|
|
230
|
+
for (const [a, b] of uncovered) {
|
|
231
|
+
for (let i = a; i + W <= b && !progress; i++) {
|
|
232
|
+
if (indexOf(fc, query.subarray(i, i + W), 0) >= 0) progress = true;
|
|
233
|
+
}
|
|
234
|
+
if (progress) break;
|
|
235
|
+
}
|
|
236
|
+
if (!progress) {
|
|
237
|
+
// THE BRAKE, MADE VISIBLE. The reasoner declines a step that carries
|
|
238
|
+
// none of the material the grounding left uncovered — the drift the
|
|
239
|
+
// extension tests pin. A refusal that leaves no trace is the kind of
|
|
240
|
+
// silent cut AGENTS §6 forbids: the rationale is where a reader learns
|
|
241
|
+
// that an extension was declined for want of question material, and
|
|
242
|
+
// where the next person sees why the chain stopped here. Measured with
|
|
243
|
+
// the check disabled, test/110 and test/116 fail — so this brake is the
|
|
244
|
+
// only thing keeping the extension honest until the pivot reports its
|
|
245
|
+
// own accounted spans and the ladder can judge it instead.
|
|
246
|
+
const left = uncovered.reduce((n, [a, b]) => n + (b - a), 0);
|
|
247
|
+
ctx.trace?.step(
|
|
248
|
+
"pivotRefused",
|
|
249
|
+
[rItem(cur, "answer"), rItem(query, "query")],
|
|
250
|
+
uncovered.map(([a, b]) => rItem(query.subarray(a, b), "uncovered")),
|
|
251
|
+
`the step carries none of the question material the grounding left ` +
|
|
252
|
+
`uncovered (${left} byte(s) in ${uncovered.length} span(s)) — refused`,
|
|
253
|
+
);
|
|
254
|
+
break;
|
|
255
|
+
}
|
|
256
|
+
}
|
|
257
|
+
if (ctx.meter) ctx.meter.pivotSteps++;
|
|
203
258
|
t ??= ctx.trace?.enter("reason", [rItem(startedFrom, "grounded")]);
|
|
204
259
|
ctx.trace?.step(
|
|
205
260
|
"pivotStep",
|
package/src/mind/traverse.ts
CHANGED
|
@@ -765,7 +765,15 @@ export function chooseNext(
|
|
|
765
765
|
// for the prevCount calls in the loop above, never for extra rItemShort
|
|
766
766
|
// byte-reads.
|
|
767
767
|
if (ctx.trace) {
|
|
768
|
-
|
|
768
|
+
// A BOUNDED SAMPLE, AND THE COUNT. The step used to carry EVERY candidate
|
|
769
|
+
// it weighed — measured on the trained store, 1559 out-items in one step
|
|
770
|
+
// (hubBound's own size) and 1082 in another (the hub's degree). The
|
|
771
|
+
// rationale's job is to explain the CHOICE, and the count is what says how
|
|
772
|
+
// wide the field was; the declared candidate budget (`recallQueryK`) is what
|
|
773
|
+
// bounds the sample, so no number is invented here.
|
|
774
|
+
const others = capped
|
|
775
|
+
.filter((c) => c !== best)
|
|
776
|
+
.slice(0, ctx.cfg.rationaleSampleK);
|
|
769
777
|
ctx.trace.step(
|
|
770
778
|
"disambiguate",
|
|
771
779
|
[rItemShort(ctx, best, "halo-evidence", bestSupport)],
|
package/src/mind/types.ts
CHANGED
|
@@ -65,12 +65,38 @@ export interface GraphSearchHost {
|
|
|
65
65
|
starts: ReadonlySet<number>;
|
|
66
66
|
};
|
|
67
67
|
chooseNext?(node: number): number | undefined;
|
|
68
|
+
/** The boundary positions of `bytes` under the engine's ONE boundary rule
|
|
69
|
+
* (geometry.ts's `contentBoundaries`), or undefined when the host has no
|
|
70
|
+
* space to ask. The join's key is an entity plus a prefix of the tail, and
|
|
71
|
+
* the prefix that names a stored relation ENDS on one of these boundaries —
|
|
72
|
+
* measured, 5 of 5 accepted keys over four join-firing queries, where the
|
|
73
|
+
* byte-by-byte scan spent 153 probes for 14 boundaries. Boundaries are
|
|
74
|
+
* content-defined and STABLE under prefix extension, which is why a corpus
|
|
75
|
+
* key's end is a boundary of the query's own fold of the same bytes. */
|
|
76
|
+
contentCuts?(bytes: Uint8Array): readonly number[];
|
|
68
77
|
/** The admission predicate — `traverse.ts`'s `leadsSomewhere`, its ONE
|
|
69
78
|
* definition: does this node bear an edge or a halo? Optional, so a bare
|
|
70
79
|
* host (a raw Store and nothing else) still works; when present, the search
|
|
71
80
|
* uses it rather than re-probing the store, which keeps the predicate
|
|
72
81
|
* single-defined AND memoised on the response-scoped struct cache. */
|
|
73
82
|
leadsSomewhere?(id: number): boolean;
|
|
83
|
+
/** Report a SEARCH REFUSAL into the rationale — the channel AGENTS §6
|
|
84
|
+
* requires: a callback threaded through a call chain must FEED the
|
|
85
|
+
* rationale, the way `GraphSearch`'s `onDerivation` feeds `traceDerivation`,
|
|
86
|
+
* never a channel of its own. Optional, so a bare host stays silent rather
|
|
87
|
+
* than crashing. */
|
|
88
|
+
reportSearch?(
|
|
89
|
+
name: string,
|
|
90
|
+
parts: ReadonlyArray<Uint8Array>,
|
|
91
|
+
note: string,
|
|
92
|
+
): void;
|
|
93
|
+
/** The CANONICAL resolver ({@link canonResolve}), optional like
|
|
94
|
+
* {@link leadsSomewhere}. The store's keys were written through the
|
|
95
|
+
* canonical fold, so a fact's `Gustaf Molander` and the deposited
|
|
96
|
+
* `gustaf molander` are the SAME node (measured inside a response: the
|
|
97
|
+
* canonical resolver maps the surface form to the deposited node while a raw
|
|
98
|
+
* resolve returns null). A bare host falls back to the plain probe. */
|
|
99
|
+
canonResolve?(bytes: Uint8Array): number | null;
|
|
74
100
|
}
|
|
75
101
|
|
|
76
102
|
// ═══════════════════════════════════════════════════════════════════════════
|