@hviana/sema 0.8.0 → 0.8.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +22 -1
- package/DATASETS.md +1 -1
- package/dist/example/train_base/config.js +2 -2
- package/dist/example/train_base/corpora/massive.js +1 -1
- package/dist/example/train_base/readers.js +1 -1
- package/dist/src/config.d.ts +17 -0
- package/dist/src/config.js +18 -0
- package/dist/src/geometry.d.ts +10 -10
- package/dist/src/geometry.js +25 -24
- package/dist/src/meter.d.ts +29 -12
- package/dist/src/meter.js +58 -14
- package/dist/src/mind/attention.js +12 -12
- package/dist/src/mind/bridge.d.ts +8 -8
- package/dist/src/mind/bridge.js +33 -32
- package/dist/src/mind/corpus.d.ts +40 -0
- package/dist/src/mind/corpus.js +149 -0
- package/dist/src/mind/graph-search.d.ts +7 -8
- package/dist/src/mind/graph-search.js +244 -32
- package/dist/src/mind/index.d.ts +3 -1
- package/dist/src/mind/index.js +1 -0
- package/dist/src/mind/junction.d.ts +1 -1
- package/dist/src/mind/junction.js +8 -8
- package/dist/src/mind/learning.js +36 -35
- package/dist/src/mind/match.d.ts +8 -3
- package/dist/src/mind/match.js +156 -71
- package/dist/src/mind/mechanisms/cast.js +18 -2
- package/dist/src/mind/mechanisms/cover.js +19 -12
- package/dist/src/mind/mechanisms/prefix-completion.js +24 -24
- package/dist/src/mind/mechanisms/recall.js +38 -40
- package/dist/src/mind/mechanisms/reference.js +16 -16
- package/dist/src/mind/mind.d.ts +61 -7
- package/dist/src/mind/mind.js +72 -2
- package/dist/src/mind/pipeline-mechanism.d.ts +10 -8
- package/dist/src/mind/pipeline-mechanism.js +25 -21
- package/dist/src/mind/pipeline.d.ts +9 -9
- package/dist/src/mind/pipeline.js +49 -29
- package/dist/src/mind/primitives.d.ts +5 -5
- package/dist/src/mind/primitives.js +5 -5
- package/dist/src/mind/reasoning.d.ts +5 -1
- package/dist/src/mind/reasoning.js +54 -1
- package/dist/src/mind/recognition.d.ts +14 -13
- package/dist/src/mind/recognition.js +23 -23
- package/dist/src/mind/resonance.js +21 -21
- package/dist/src/mind/traverse.d.ts +54 -52
- package/dist/src/mind/traverse.js +83 -73
- package/dist/src/mind/types.d.ts +26 -4
- package/dist/src/store.d.ts +12 -12
- package/dist/src/store.js +12 -12
- package/docs/INDEX.md +2 -2
- package/docs/architecture/exact-vs-approximate.md +2 -1
- package/docs/architecture/fold-contract.md +1 -1
- package/docs/failures/tempting-but-wrong.md +33 -5
- package/docs/harness/gates.md +7 -7
- package/example/train_base/config.ts +2 -2
- package/example/train_base/corpora/massive.ts +1 -1
- package/example/train_base/readers.ts +1 -1
- package/jsr.json +1 -1
- package/package.json +1 -1
- package/src/config.ts +35 -0
- package/src/geometry.ts +25 -24
- package/src/meter.ts +61 -14
- package/src/mind/attention.ts +12 -12
- package/src/mind/bridge.ts +33 -32
- package/src/mind/corpus.ts +202 -0
- package/src/mind/graph-search.ts +261 -31
- package/src/mind/index.ts +8 -1
- package/src/mind/junction.ts +8 -8
- package/src/mind/learning.ts +36 -35
- package/src/mind/match.ts +163 -73
- package/src/mind/mechanisms/cast.ts +17 -1
- package/src/mind/mechanisms/cover.ts +18 -12
- package/src/mind/mechanisms/prefix-completion.ts +24 -24
- package/src/mind/mechanisms/recall.ts +38 -40
- package/src/mind/mechanisms/reference.ts +16 -16
- package/src/mind/mind.ts +129 -7
- package/src/mind/pipeline-mechanism.ts +25 -21
- package/src/mind/pipeline.ts +63 -38
- package/src/mind/primitives.ts +5 -5
- package/src/mind/reasoning.ts +55 -0
- package/src/mind/recognition.ts +23 -23
- package/src/mind/resonance.ts +21 -21
- package/src/mind/traverse.ts +83 -73
- package/src/mind/types.ts +30 -4
- package/src/store.ts +20 -20
- package/test/08-storage.test.mjs +1 -1
- package/test/100-complete-grounding-trace.test.mjs +109 -0
- package/test/101-alignment-gap-bound.test.mjs +106 -0
- package/test/102-production-composes-at-scale.test.mjs +110 -0
- package/test/103-alignment-gap-budget.test.mjs +89 -0
- package/test/104-composition-is-reported.test.mjs +90 -0
- package/test/105-derive-through-reports-its-refusal.test.mjs +113 -0
- package/test/106-the-join-fires.test.mjs +94 -0
- package/test/107-the-join-is-counted.test.mjs +81 -0
- package/test/108-the-join-chains.test.mjs +78 -0
- package/test/109-the-pivot-is-counted.test.mjs +60 -0
- package/test/110-the-reasoner-stops-when-the-question-is-answered.test.mjs +91 -0
- package/test/111-the-cover-assembly-is-counted.test.mjs +74 -0
- package/test/112-the-exploration-does-not-grow-with-the-hub.test.mjs +89 -0
- package/test/113-the-rationale-payload-is-bounded.test.mjs +84 -0
- package/test/114-alignment-budget-is-per-sweep.test.mjs +93 -0
- package/test/116-the-extension-is-gated-by-the-pipelines-own-remainder.test.mjs +100 -0
- package/test/117-corpus-search.test.mjs +171 -0
- package/test/14-scaling.test.mjs +10 -7
- package/test/35-prefix-edge.test.mjs +1 -1
- package/test/40-choosenext-scale-guard.test.mjs +16 -17
- package/test/56-bridge-identity-admission.test.mjs +6 -6
- package/test/70-prefix-completion.test.mjs +4 -3
- package/test/72-prefix-candidate-supply.test.mjs +3 -3
- package/test/73-scaffolding-only-bridge-abstains.test.mjs +6 -6
- package/test/75-multiturn-context-optimisation.test.mjs +5 -5
- package/test/76-reference-binding.test.mjs +6 -1
- package/test/84-composed-answer-honesty.test.mjs +5 -6
- package/test/88-dependency-footprint.test.mjs +1 -1
- package/test/89-completion-recursion.test.mjs +47 -19
- package/test/90-connector-read-cap.test.mjs +10 -8
- package/test/93-regime-prediction.test.mjs +10 -10
- package/test/94-cross-region-budget.test.mjs +2 -2
- package/test/95-wide-resonance-removed.test.mjs +8 -7
- package/test/96-bytes-walk-termination.test.mjs +3 -3
package/src/mind/match.ts
CHANGED
|
@@ -38,7 +38,7 @@ import {
|
|
|
38
38
|
identityBar,
|
|
39
39
|
significanceBar,
|
|
40
40
|
} from "../geometry.js";
|
|
41
|
-
import { bytesEqual, indexOf } from "../bytes.js";
|
|
41
|
+
import { bytesEqual, indexOf, latin1 } from "../bytes.js";
|
|
42
42
|
import type { MindContext } from "./types.js";
|
|
43
43
|
import { chainReach, leafIdRun } from "./canonical.js";
|
|
44
44
|
import { foldTree, gistOf, perceive, read, resolve } from "./primitives.js";
|
|
@@ -54,6 +54,7 @@ import {
|
|
|
54
54
|
sharedReachMemo,
|
|
55
55
|
} from "./traverse.js";
|
|
56
56
|
import { recognise, segment } from "./recognition.js";
|
|
57
|
+
import { rItem } from "./trace.js";
|
|
57
58
|
import type { Site } from "./graph-search.js";
|
|
58
59
|
|
|
59
60
|
// ═══════════════════════════════════════════════════════════════════════════
|
|
@@ -319,12 +320,12 @@ export function alignGraded(
|
|
|
319
320
|
//
|
|
320
321
|
// "these bytes occupy a place the corpus keeps open" (bind).
|
|
321
322
|
//
|
|
322
|
-
// Both arrive as unaligned residue.
|
|
323
|
-
//
|
|
324
|
-
//
|
|
325
|
-
//
|
|
326
|
-
//
|
|
327
|
-
//
|
|
323
|
+
// Both arrive as unaligned residue. That single missing distinction is why the
|
|
324
|
+
// substitution bridge refuses on `attestedQ`, why the cover charges PASS over a
|
|
325
|
+
// slot, and why CAST reads a filler as noise rather than as the variable it is.
|
|
326
|
+
// The family below supplies it, and it lives HERE — not in any mechanism —
|
|
327
|
+
// because it is the ordinary (matcher, projection, gate) triple of
|
|
328
|
+
// match-project.md with its three parts in their proper places:
|
|
328
329
|
//
|
|
329
330
|
// matcher alignAround + contractGap + frameSlots — bytes only, no
|
|
330
331
|
// projection, no licence. SAFE FOR EVERY CONSUMER: knowing a
|
|
@@ -360,9 +361,14 @@ export interface AlignGap {
|
|
|
360
361
|
|
|
361
362
|
/** Extend a seed match (query offset qo ↔ candidate offset co) to its maximal
|
|
362
363
|
* common run, then walk outward in both directions collecting further common
|
|
363
|
-
* runs of at least W bytes across
|
|
364
|
-
*
|
|
365
|
-
*
|
|
364
|
+
* runs of at least W bytes across mismatch gaps. Each gap's LENGTH is the
|
|
365
|
+
* pair's own extent (a gap cannot be longer than the bytes it spans) and the
|
|
366
|
+
* sweep's WORK is proportional to the bytes a run spans (the context's windows
|
|
367
|
+
* are indexed once, then the query's are walked) — the arity bound
|
|
368
|
+
* (`chainReach`) used to cap BOTH, and truncated every learned frame whose
|
|
369
|
+
* slot was longer. Each sweep owns its own budget, so an exhausted right
|
|
370
|
+
* sweep never starves the left one. Returns the matched query spans and the
|
|
371
|
+
* mismatch pairs between consecutive runs.
|
|
366
372
|
*
|
|
367
373
|
* This is the SEEDED aligner, distinct from {@link alignRuns}: that one finds
|
|
368
374
|
* every run two structures share anywhere (a weave), this one reads two
|
|
@@ -382,7 +388,21 @@ export function alignAround(
|
|
|
382
388
|
co: number,
|
|
383
389
|
): { matched: Array<[number, number]>; gaps: AlignGap[] } {
|
|
384
390
|
const W = ctx.space.maxGroup;
|
|
385
|
-
|
|
391
|
+
// THE GAP LENGTH IS THE PAIR'S OWN EXTENT; THE WORK IS BUDGETED.
|
|
392
|
+
//
|
|
393
|
+
// The sweep walks (queryGap, contextGap) pairs by ASCENDING total, so reaching
|
|
394
|
+
// a gap of size G costs about G²/2 pairs. Bounding the LENGTH by the write
|
|
395
|
+
// side's arity (`chainReach(W)` = 16) therefore truncated every learned frame
|
|
396
|
+
// whose slot is longer — measured: `bindReference` reported the cap at 18, 24,
|
|
397
|
+
// 30 and 36 bytes and `recall` answered with ANOTHER instance's filler — while
|
|
398
|
+
// removing the bound outright took the corpus-cost guard (test/89) from
|
|
399
|
+
// milliseconds to 68 seconds.
|
|
400
|
+
//
|
|
401
|
+
// Bounding the PAIRS keeps a call's cost constant however long the pair is,
|
|
402
|
+
// and the ascending order means an exhausted budget drops the FAR
|
|
403
|
+
// continuations and never the near ones — the same degradation recognition.ts
|
|
404
|
+
// documents for its canon budget. Length and work are different questions;
|
|
405
|
+
// this is the one place they were conflated.
|
|
386
406
|
// Maximal run around the seed.
|
|
387
407
|
let qs = qo, ss = co;
|
|
388
408
|
while (qs > 0 && ss > 0 && q[qs - 1] === c[ss - 1]) {
|
|
@@ -396,8 +416,60 @@ export function alignAround(
|
|
|
396
416
|
}
|
|
397
417
|
const matched: Array<[number, number]> = [[qs, qe]];
|
|
398
418
|
const gaps: AlignGap[] = [];
|
|
399
|
-
//
|
|
400
|
-
//
|
|
419
|
+
// THE SWEEP IS STRUCTURAL, NOT ENUMERATIVE.
|
|
420
|
+
//
|
|
421
|
+
// The criterion is unchanged: the next common run, MINIMUM TOTAL GAP, ties to
|
|
422
|
+
// the smaller query gap. What changed is how it is found. Enumerating
|
|
423
|
+
// (queryGap, contextGap) pairs by ascending total reaches a run at total t in
|
|
424
|
+
// about t²/2 pairs — and that quadratic shape, not the reach, was the cost
|
|
425
|
+
// problem: capping the pairs dropped reach (a legitimate 24-byte slot stopped
|
|
426
|
+
// being found), while leaving them uncapped cost 68 seconds on the corpus
|
|
427
|
+
// guard. Neither is the answer, because the answer is the algorithm.
|
|
428
|
+
//
|
|
429
|
+
// The context's windows are indexed ONCE, for lengths 1..W — W being the
|
|
430
|
+
// geometry's own unit of composition, so nothing is chosen here. Each step
|
|
431
|
+
// then walks the query's windows outward from the anchor: for a given query
|
|
432
|
+
// gap the nearest context gap that continues a run is one O(1) lookup, and the
|
|
433
|
+
// walk stops the moment the query gap alone exceeds the best total already
|
|
434
|
+
// found. So the work is proportional to the bytes the run SPANS. No budget,
|
|
435
|
+
// no cap, no number: a long slot is reached, and its price is already the
|
|
436
|
+
// ladder's (its bytes are unaccounted, so the search pays PASS per byte).
|
|
437
|
+
const index: Array<Map<string, number[]>> = [];
|
|
438
|
+
for (let len = 1; len <= W; len++) {
|
|
439
|
+
const m = new Map<string, number[]>();
|
|
440
|
+
for (let o = 0; o + len <= c.length; o++) {
|
|
441
|
+
const key = latin1(c.subarray(o, o + len));
|
|
442
|
+
const at = m.get(key);
|
|
443
|
+
if (at === undefined) m.set(key, [o]);
|
|
444
|
+
else at.push(o);
|
|
445
|
+
}
|
|
446
|
+
index.push(m);
|
|
447
|
+
}
|
|
448
|
+
/** Smallest listed offset at or after `from`, or -1. */
|
|
449
|
+
const fromAt = (list: number[], from: number): number => {
|
|
450
|
+
let lo = 0, hi = list.length - 1, best = -1;
|
|
451
|
+
while (lo <= hi) {
|
|
452
|
+
const mid = (lo + hi) >> 1;
|
|
453
|
+
if (list[mid] >= from) {
|
|
454
|
+
best = list[mid];
|
|
455
|
+
hi = mid - 1;
|
|
456
|
+
} else lo = mid + 1;
|
|
457
|
+
}
|
|
458
|
+
return best;
|
|
459
|
+
};
|
|
460
|
+
/** Largest listed offset at or before `to`, or -1. */
|
|
461
|
+
const toAt = (list: number[], to: number): number => {
|
|
462
|
+
let lo = 0, hi = list.length - 1, best = -1;
|
|
463
|
+
while (lo <= hi) {
|
|
464
|
+
const mid = (lo + hi) >> 1;
|
|
465
|
+
if (list[mid] <= to) {
|
|
466
|
+
best = list[mid];
|
|
467
|
+
lo = mid + 1;
|
|
468
|
+
} else hi = mid - 1;
|
|
469
|
+
}
|
|
470
|
+
return best;
|
|
471
|
+
};
|
|
472
|
+
/** Length of the common run STARTING at (qi, si). */
|
|
401
473
|
const runLenAt = (qi: number, si: number): number => {
|
|
402
474
|
let n = 0;
|
|
403
475
|
while (qi + n < q.length && si + n < c.length && q[qi + n] === c[si + n]) {
|
|
@@ -405,63 +477,80 @@ export function alignAround(
|
|
|
405
477
|
}
|
|
406
478
|
return n;
|
|
407
479
|
};
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
|
|
421
|
-
|
|
480
|
+
/** Length of the common run ENDING at (qi, si). */
|
|
481
|
+
const runLenBefore = (qi: number, si: number): number => {
|
|
482
|
+
let n = 0;
|
|
483
|
+
while (n < qi && n < si && q[qi - 1 - n] === c[si - 1 - n]) n++;
|
|
484
|
+
return n;
|
|
485
|
+
};
|
|
486
|
+
/** The next run outward from an anchor, or null when the bytes run out. */
|
|
487
|
+
const nextRun = (
|
|
488
|
+
qi: number,
|
|
489
|
+
si: number,
|
|
490
|
+
forward: boolean,
|
|
491
|
+
): { gq: number; gs: number; n: number } | null => {
|
|
492
|
+
const qLim = forward ? q.length - qi : qi;
|
|
493
|
+
let best: { gq: number; gs: number; n: number } | null = null;
|
|
494
|
+
for (let gq = 0; gq < qLim; gq++) {
|
|
495
|
+
// No later query gap can beat a total already found.
|
|
496
|
+
if (best !== null && gq > best.gq + best.gs) break;
|
|
497
|
+
const left = qLim - gq;
|
|
498
|
+
// A run of >= W bytes, or — when the query itself ends inside one window —
|
|
499
|
+
// the run that REACHES that end. Exactly the acceptance the sweep had.
|
|
500
|
+
const lens = left >= W ? [W] : [left];
|
|
501
|
+
for (const len of lens) {
|
|
502
|
+
const key = latin1(
|
|
503
|
+
q.subarray(
|
|
504
|
+
forward ? qi + gq : qi - gq - len,
|
|
505
|
+
forward ? qi + gq + len : qi - gq,
|
|
506
|
+
),
|
|
507
|
+
);
|
|
508
|
+
const list = index[len - 1].get(key);
|
|
509
|
+
if (list === undefined) continue;
|
|
510
|
+
const o = forward ? fromAt(list, si) : toAt(list, si - len);
|
|
511
|
+
if (o < 0) continue;
|
|
512
|
+
const n = forward
|
|
513
|
+
? runLenAt(qi + gq, o)
|
|
514
|
+
: runLenBefore(qi - gq, o + len);
|
|
515
|
+
if (n < 1) continue;
|
|
516
|
+
if (
|
|
517
|
+
forward ? n >= W || qi + gq + n === q.length : n >= W || n === qi - gq
|
|
518
|
+
) {
|
|
519
|
+
const gs = forward ? o - si : si - len - o;
|
|
520
|
+
if (best === null || gq + gs < best.gq + best.gs) {
|
|
521
|
+
best = { gq, gs, n };
|
|
422
522
|
}
|
|
423
|
-
matched.push([qi + gq, qi + gq + n]);
|
|
424
|
-
qi = qi + gq + n;
|
|
425
|
-
si = si + gs + n;
|
|
426
|
-
found = true;
|
|
427
523
|
break;
|
|
428
524
|
}
|
|
429
525
|
}
|
|
430
526
|
}
|
|
431
|
-
|
|
527
|
+
return best;
|
|
528
|
+
};
|
|
529
|
+
// RIGHT sweep.
|
|
530
|
+
let qi = qe, si = se;
|
|
531
|
+
for (;;) {
|
|
532
|
+
const step = nextRun(qi, si, true);
|
|
533
|
+
if (step === null) break;
|
|
534
|
+
if (step.gq > 0 || step.gs > 0) {
|
|
535
|
+
gaps.push({ qs: qi, qe: qi + step.gq, cs: si, ce: si + step.gs });
|
|
536
|
+
}
|
|
537
|
+
matched.push([qi + step.gq, qi + step.gq + step.n]);
|
|
538
|
+
qi = qi + step.gq + step.n;
|
|
539
|
+
si = si + step.gs + step.n;
|
|
432
540
|
}
|
|
433
|
-
// LEFT sweep (mirror)
|
|
541
|
+
// LEFT sweep (mirror): an independent walk, so an exhausted right side can
|
|
542
|
+
// never starve it (pinned by test/114).
|
|
434
543
|
qi = qs;
|
|
435
544
|
si = ss;
|
|
436
545
|
for (;;) {
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
|
|
441
|
-
if (gs > reachCap) continue;
|
|
442
|
-
if (qi - gq <= 0 || si - gs <= 0) continue;
|
|
443
|
-
// Run ENDING at (qi - gq, si - gs).
|
|
444
|
-
let n = 0;
|
|
445
|
-
while (
|
|
446
|
-
n < qi - gq && n < si - gs &&
|
|
447
|
-
q[qi - gq - 1 - n] === c[si - gs - 1 - n]
|
|
448
|
-
) {
|
|
449
|
-
n++;
|
|
450
|
-
}
|
|
451
|
-
if (n >= W || n === qi - gq) {
|
|
452
|
-
if (n === 0) continue;
|
|
453
|
-
if (gq > 0 || gs > 0) {
|
|
454
|
-
gaps.push({ qs: qi - gq, qe: qi, cs: si - gs, ce: si });
|
|
455
|
-
}
|
|
456
|
-
matched.push([qi - gq - n, qi - gq]);
|
|
457
|
-
qi = qi - gq - n;
|
|
458
|
-
si = si - gs - n;
|
|
459
|
-
found = true;
|
|
460
|
-
break;
|
|
461
|
-
}
|
|
462
|
-
}
|
|
546
|
+
const step = nextRun(qi, si, false);
|
|
547
|
+
if (step === null) break;
|
|
548
|
+
if (step.gq > 0 || step.gs > 0) {
|
|
549
|
+
gaps.push({ qs: qi - step.gq, qe: qi, cs: si - step.gs, ce: si });
|
|
463
550
|
}
|
|
464
|
-
|
|
551
|
+
matched.push([qi - step.gq - step.n, qi - step.gq]);
|
|
552
|
+
qi = qi - step.gq - step.n;
|
|
553
|
+
si = si - step.gs - step.n;
|
|
465
554
|
}
|
|
466
555
|
return { matched, gaps };
|
|
467
556
|
}
|
|
@@ -1168,24 +1257,25 @@ export async function project(
|
|
|
1168
1257
|
|
|
1169
1258
|
// ── The span-shape family ───────────────────────────────────────────────────
|
|
1170
1259
|
//
|
|
1171
|
-
// "Is this answer drawn from this context?" has TWO formally distinct
|
|
1172
|
-
//
|
|
1173
|
-
//
|
|
1174
|
-
//
|
|
1175
|
-
//
|
|
1176
|
-
//
|
|
1177
|
-
//
|
|
1178
|
-
//
|
|
1179
|
-
//
|
|
1180
|
-
//
|
|
1181
|
-
//
|
|
1260
|
+
// "Is this answer drawn from this context?" has TWO formally distinct readings,
|
|
1261
|
+
// and the pair plus the anchor classifier built on them are SHARED machinery —
|
|
1262
|
+
// extraction proposes span-shaped exemplars with them, the shared
|
|
1263
|
+
// `Precomputed.spanShapedOf` container computes them, and fusion (reasoning.ts)
|
|
1264
|
+
// gates on the strict one. They lived inside mechanisms/extraction.ts, so
|
|
1265
|
+
// `pipeline-mechanism.ts` and `reasoning.ts` both had to import back OUT of a
|
|
1266
|
+
// specific mechanism — an inversion the mechanism market forbids: the shared
|
|
1267
|
+
// contract may not depend on any one mechanism (mechanism-market.md), and a
|
|
1268
|
+
// shared matcher belongs to this family (match-project.md), never to a
|
|
1269
|
+
// mechanism's private helpers. Deleting extraction must not break the shared
|
|
1270
|
+
// container, so they live here.
|
|
1182
1271
|
//
|
|
1183
1272
|
// • isSpanShaped — the OPEN reading (sparse in-order embedding).
|
|
1184
1273
|
// • containsSpan — the STRICT reading (contiguous run or resolved node).
|
|
1185
1274
|
// • skillExemplar — classify one anchor into (context, answer) using them.
|
|
1186
1275
|
//
|
|
1187
|
-
// The two readings are NOT interchangeable;
|
|
1188
|
-
// and each function's own doc states what breaks if it is
|
|
1276
|
+
// The two readings are NOT interchangeable; match-project.md pins the
|
|
1277
|
+
// distinction and each function's own doc states what breaks if it is
|
|
1278
|
+
// substituted.
|
|
1189
1279
|
|
|
1190
1280
|
/** Check whether an anchor is a span-shaped skill exemplar: it represents a
|
|
1191
1281
|
* fact whose context and answer together form a span-in-context pattern.
|
|
@@ -615,7 +615,23 @@ export async function counterfactualTransfer(
|
|
|
615
615
|
fwd !== null && indexOf(answer, fwd, 0) < 0 &&
|
|
616
616
|
!restatesQuery(query, fwd)
|
|
617
617
|
) {
|
|
618
|
-
|
|
618
|
+
// THROUGH THE SHARED JOINER, not a bare concatenation.
|
|
619
|
+
//
|
|
620
|
+
// `joinWithBridge` is the composition step every out-of-search assembly
|
|
621
|
+
// shares (multi-topic fusion, CAST's substitution and comparison): it
|
|
622
|
+
// asks the corpus for a learnt connector between the pieces and, on a
|
|
623
|
+
// miss, joins them BARE **and says so** — the `bridgeMiss` step (see
|
|
624
|
+
// resonance.ts). This site bypassed it, and that is the whole of the
|
|
625
|
+
// gluing the study measured: `"Steel is hard"` + `"wet"` came back as
|
|
626
|
+
// `"hardwet"`, `"eva director father"` + `"The father of…"` as
|
|
627
|
+
// `"fatherThe"` — compositions no rationale could show, because the one
|
|
628
|
+
// step that made them left no trace.
|
|
629
|
+
//
|
|
630
|
+
// Routing it through the shared joiner is the instrumentation fix that
|
|
631
|
+
// comes first: a bare join stays possible (the house rule is "joined
|
|
632
|
+
// bare, never silent") but it is now VISIBLE, and an attested connector
|
|
633
|
+
// is used when the corpus has one.
|
|
634
|
+
answer = await joinWithBridge(ctx, answer, fwd);
|
|
619
635
|
}
|
|
620
636
|
ctx.trace?.step(
|
|
621
637
|
"projectCounterfactual",
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
// Cover consumes recognition directly (its axioms are the query's own
|
|
5
5
|
// decomposition) plus the computed spans any parse()-bearing mechanism
|
|
6
6
|
// contributed: computed spans MASK colliding recognised sites and enter the
|
|
7
|
-
// search at zero cost ("computation always wins",
|
|
7
|
+
// search at zero cost ("computation always wins", alu.md) — which is also why
|
|
8
8
|
// cover runs FIRST in defaultMechanisms: a computed-backed cover becomes a
|
|
9
9
|
// near-zero-cost incumbent that prunes the other mechanisms through the
|
|
10
10
|
// ordinary admissible-floor check, with no extension special-case anywhere.
|
|
@@ -75,28 +75,30 @@ export async function resolveConnectors(
|
|
|
75
75
|
if (query === undefined || ctx.answeredSpans.length === 0) return true;
|
|
76
76
|
const continuations = ctx.store.nextFirst(s.payload, hubBound(ctx));
|
|
77
77
|
return !continuations.some((answer) => {
|
|
78
|
-
// PREFIX-CAPPED (
|
|
79
|
-
// occur INSIDE it, so read one byte past the query's length —
|
|
80
|
-
// detect the overflow — and reject without reconstructing the
|
|
81
|
-
// The `+ 1` is what makes the test exact rather than a
|
|
82
|
-
// result of exactly `query.length + 1` bytes is known to
|
|
83
|
-
// and anything shorter is the candidate's COMPLETE
|
|
84
|
-
// substring test below is the same test as before.
|
|
85
|
-
// probe bridge.ts:256 already uses.)
|
|
78
|
+
// PREFIX-CAPPED (bounded-reads.md): a candidate longer than the query
|
|
79
|
+
// cannot occur INSIDE it, so read one byte past the query's length —
|
|
80
|
+
// enough to detect the overflow — and reject without reconstructing the
|
|
81
|
+
// rest. The `+ 1` is what makes the test exact rather than a
|
|
82
|
+
// truncation: a result of exactly `query.length + 1` bytes is known to
|
|
83
|
+
// be too long, and anything shorter is the candidate's COMPLETE
|
|
84
|
+
// content, so the substring test below is the same test as before. (The
|
|
85
|
+
// same overflow probe bridge.ts:256 already uses.)
|
|
86
86
|
//
|
|
87
87
|
// This loop runs up to hubBound(ctx) = √N reads PER SITE, and only on a
|
|
88
88
|
// multi-turn response — `answeredSpans` is empty for a plain respond(),
|
|
89
|
-
// so the probe does not execute there.
|
|
89
|
+
// so the probe does not execute there. The cap cannot reduce the read
|
|
90
90
|
// COUNT — only a semantic change to the "already answered" test could —
|
|
91
91
|
// but it bounds each read by the query instead of by the corpus, which
|
|
92
|
-
// is what
|
|
93
|
-
// reads 4 bytes per candidate instead of the ~231 it
|
|
92
|
+
// is what bounded-reads.md asks for and what rescues a SHORT query: at
|
|
93
|
+
// 3 bytes this reads 4 bytes per candidate instead of the ~231 it
|
|
94
|
+
// averaged before.
|
|
94
95
|
const bytes = read(ctx, answer, query.length + 1);
|
|
95
96
|
return bytes.length <= query.length && indexOf(query, bytes, 0) >= 0;
|
|
96
97
|
});
|
|
97
98
|
});
|
|
98
99
|
const bridgePair = async (l: number, r: number) => {
|
|
99
100
|
if (l === r || links.has(l + "," + r)) return;
|
|
101
|
+
if (ctx.meter) ctx.meter.coverBridges++;
|
|
100
102
|
const link = await bridge(ctx, read(ctx, l), read(ctx, r));
|
|
101
103
|
if (link !== null) links.set(l + "," + r, link);
|
|
102
104
|
};
|
|
@@ -133,6 +135,10 @@ export async function resolveConnectors(
|
|
|
133
135
|
// plus one W-quantum of glue per joint — pass that allowance so the
|
|
134
136
|
// bridge's phrase-scale cap admits the whole learnt run.
|
|
135
137
|
const allowance = middleBytes + (m + 1) * W;
|
|
138
|
+
if (ctx.meter) {
|
|
139
|
+
ctx.meter.coverBridges++;
|
|
140
|
+
ctx.meter.coverAllowanceBytes += allowance;
|
|
141
|
+
}
|
|
136
142
|
const interior = await bridge(
|
|
137
143
|
ctx,
|
|
138
144
|
first.bytes,
|
|
@@ -1,15 +1,15 @@
|
|
|
1
1
|
// mechanisms/prefix-completion.ts — Grounding a query that IS the opening of a
|
|
2
2
|
// trained form (Grounding V).
|
|
3
3
|
//
|
|
4
|
-
// A MECHANISM, NOT A TIER.
|
|
5
|
-
//
|
|
6
|
-
// from, where placement rather than the cost ladder decided.
|
|
4
|
+
// A MECHANISM, NOT A TIER. This used to run inside recall's refusal path, in a
|
|
5
|
+
// fixed if-chain that first-match-wins — the shape CAST was refactored away
|
|
6
|
+
// from, where placement rather than the cost ladder decided. Its claim is
|
|
7
7
|
// maximal (every query byte literally matched, from offset zero, against a
|
|
8
8
|
// trained form) at one STEP, so as a market candidate it competes honestly and
|
|
9
|
-
// the decider weighs it like everything else.
|
|
9
|
+
// the decider weighs it like everything else. It is registered LAST: recall's
|
|
10
10
|
// exact self-match makes an IDENTITY claim about the query while this makes a
|
|
11
|
-
// CONTAINMENT one, and on an exact grade tie the identity claim is the
|
|
12
|
-
//
|
|
11
|
+
// CONTAINMENT one, and on an exact grade tie the identity claim is the stronger
|
|
12
|
+
// evidence — the same ordering exact-vs-approximate.md's ladders use.
|
|
13
13
|
//
|
|
14
14
|
// Its SUPPLY moved too, and further: `formsOpenedBy` (traverse.ts) answers a
|
|
15
15
|
// question about the STORE — "which trained forms does this byte run open?" —
|
|
@@ -49,12 +49,12 @@
|
|
|
49
49
|
//
|
|
50
50
|
// So this is a RETRIEVABILITY gap, not a semantic one, and the ANN is the wrong
|
|
51
51
|
// instrument for it: a proper prefix's gist cannot rank its own continuation.
|
|
52
|
-
// The repair is CONTENT-ADDRESSED (
|
|
53
|
-
// the leaf-id WINDOW index the write side already maintains
|
|
54
|
-
// trained forms does this byte run open?" in a bounded √N
|
|
55
|
-
// mechanism's first supply.
|
|
56
|
-
// second, for prefixes long enough that the gist still
|
|
57
|
-
// read, never re-issued.
|
|
52
|
+
// The repair is CONTENT-ADDRESSED (exact-vs-approximate.md) — `formsOpenedBy`
|
|
53
|
+
// (traverse.ts) reads the leaf-id WINDOW index the write side already maintains
|
|
54
|
+
// and answers "which trained forms does this byte run open?" in a bounded √N
|
|
55
|
+
// walk. That is this mechanism's first supply. The response's memoised top-k
|
|
56
|
+
// `resonance()` is the second, for prefixes long enough that the gist still
|
|
57
|
+
// ranks the form; it is read, never re-issued.
|
|
58
58
|
//
|
|
59
59
|
// AN EXHAUSTIVE ANN LIST IS NOT A SUPPLY HERE, AND WAS REMOVED. This tier once
|
|
60
60
|
// read `Precomputed.wideResonance()` — a full-index `resonate(guide, √N,
|
|
@@ -273,20 +273,20 @@ export const prefixMechanism: PipelineMechanism = {
|
|
|
273
273
|
return STEP;
|
|
274
274
|
},
|
|
275
275
|
async run(ctx, query, pre) {
|
|
276
|
-
// ONE SUPPLY PASS, not a two-tier `??`.
|
|
276
|
+
// ONE SUPPLY PASS, not a two-tier `??`. The window index (exact,
|
|
277
277
|
// content-addressed) and the response's memoised top-k (approximate) are
|
|
278
|
-
// concatenated and the three guards decide ONCE over the union.
|
|
278
|
+
// concatenated and the three guards decide ONCE over the union. A
|
|
279
279
|
// first-then-fallback chain would let the APPROXIMATE tier override the
|
|
280
|
-
// EXACT one (
|
|
281
|
-
// returns null and the fallback re-runs the guards
|
|
282
|
-
// alone — which, seeing only one of the two forms,
|
|
283
|
-
// precisely the disagreement-suppression guard 3
|
|
284
|
-
// is the exact tier's ambiguity being washed away
|
|
285
|
-
// Evaluating the union means a disagreement the
|
|
286
|
-
// be hidden by what the ANN happens to rank.
|
|
287
|
-
// response's ONE memoised top-k (
|
|
288
|
-
// path on the queries where this mechanism fires,
|
|
289
|
-
// a second index scan.
|
|
280
|
+
// EXACT one (exact-vs-approximate.md): when formsOpenedBy finds two
|
|
281
|
+
// continuations, guard 3 returns null and the fallback re-runs the guards
|
|
282
|
+
// on resonance's top-k alone — which, seeing only one of the two forms,
|
|
283
|
+
// would voice it. That is precisely the disagreement-suppression guard 3
|
|
284
|
+
// exists to prevent, and it is the exact tier's ambiguity being washed away
|
|
285
|
+
// by the approximate tier. Evaluating the union means a disagreement the
|
|
286
|
+
// window index saw can never be hidden by what the ANN happens to rank. The
|
|
287
|
+
// ANN read is the response's ONE memoised top-k (memoization.md), already
|
|
288
|
+
// paid by recall's refusal path on the queries where this mechanism fires,
|
|
289
|
+
// so reading it here is not a second index scan.
|
|
290
290
|
const ids = [
|
|
291
291
|
...formsOpenedBy(ctx, query),
|
|
292
292
|
...(await pre.resonance()).map((h) => h.id),
|
|
@@ -236,24 +236,22 @@ export async function recallByResonance(
|
|
|
236
236
|
}
|
|
237
237
|
}
|
|
238
238
|
|
|
239
|
-
// The query-relative grounding fraction, shared by tiers 2–4 — gated on
|
|
240
|
-
//
|
|
241
|
-
//
|
|
242
|
-
//
|
|
243
|
-
//
|
|
244
|
-
//
|
|
245
|
-
//
|
|
246
|
-
//
|
|
247
|
-
//
|
|
248
|
-
//
|
|
249
|
-
//
|
|
250
|
-
// √(lenG/lenQ)
|
|
251
|
-
//
|
|
252
|
-
//
|
|
253
|
-
//
|
|
254
|
-
//
|
|
255
|
-
// subtract the significance bar (3/√D, §8.3) before converting. Derived
|
|
256
|
-
// from the existing bars; never tuned.
|
|
239
|
+
// The query-relative grounding fraction, shared by tiers 2–4 — gated on the
|
|
240
|
+
// FRACTION OF THE QUERY the grounding explains, not the raw cosine. Root
|
|
241
|
+
// gists are unit vectors, but their magnitudes are recoverable from the byte
|
|
242
|
+
// lengths (‖·‖ = √len under the linear fold): cos = shared/√(lenQ·lenG), so
|
|
243
|
+
// shared/lenQ = cos·√(lenG/lenQ). The raw cosine punished honest containment
|
|
244
|
+
// — a query fully inside a longer grounded answer scored √(lenQ/lenG) and was
|
|
245
|
+
// refused — and let a long answer sharing only scaffolding pass; the
|
|
246
|
+
// query-relative fraction measures exactly what the reach bar means: how much
|
|
247
|
+
// of THE QUERY the store accounts for. Chance similarity survives the length
|
|
248
|
+
// conversion AMPLIFIED: the same √(lenG/lenQ) factor that converts an honest
|
|
249
|
+
// shared fraction into a query-relative one multiplies the estimator/chance
|
|
250
|
+
// floor too, so a long stored form (√(lenG/lenQ) ≈ 10 at 100×) lifted a
|
|
251
|
+
// noise-level cosine past the reach bar and grounded pure gibberish
|
|
252
|
+
// (observed). Only the ABOVE-CHANCE part of the similarity is evidence of
|
|
253
|
+
// shared content — subtract the significance bar (3/√D, thresholds.md) before
|
|
254
|
+
// converting. Derived from the existing bars; never tuned.
|
|
257
255
|
const sig = significanceBar(ctx.store.D);
|
|
258
256
|
const reach = reachThreshold(ctx.space.maxGroup);
|
|
259
257
|
const fracOfQuery = (cos: number, otherLen: number): number =>
|
|
@@ -411,14 +409,14 @@ export async function recallByResonance(
|
|
|
411
409
|
}
|
|
412
410
|
}
|
|
413
411
|
}
|
|
414
|
-
// 3b. Corroborated-substitution bridge — refusal-path only (bridge.ts).
|
|
415
|
-
//
|
|
416
|
-
//
|
|
417
|
-
//
|
|
418
|
-
//
|
|
419
|
-
//
|
|
420
|
-
// cluster here once made every honest refusal cost hundreds of ms
|
|
421
|
-
// of k.
|
|
412
|
+
// 3b. Corroborated-substitution bridge — refusal-path only (bridge.ts). The
|
|
413
|
+
// bridge's proposal source is the response's ONE top-k read — the same list
|
|
414
|
+
// recall already ranked above — never an exhaustive √N scan. The bridge's own
|
|
415
|
+
// candidate cap is 2·recallQueryK, so top-k proposals are exactly the budget
|
|
416
|
+
// it can consume, and every proposal is byte-verified downstream
|
|
417
|
+
// (exact-vs-approximate.md). Reuse the memoised `resonance()`; scanning every
|
|
418
|
+
// IVF cluster here once made every honest refusal cost hundreds of ms
|
|
419
|
+
// regardless of k.
|
|
422
420
|
const wideIds = async () => (await pre.resonance()).map((h) => h.id);
|
|
423
421
|
|
|
424
422
|
// Every gist-based tier has failed; before refusing, align the query
|
|
@@ -471,12 +469,12 @@ export async function recallByResonance(
|
|
|
471
469
|
// prefixCompletion runs a few lines below and carries the three guards
|
|
472
470
|
// this tier lacks — unreadable-continuation veto, sub-quantum
|
|
473
471
|
// continuation, and UNIQUENESS (distinct continuations ⇒ refuse), which
|
|
474
|
-
// is exactly what 4,300 competing values must trip.
|
|
475
|
-
//
|
|
476
|
-
//
|
|
477
|
-
// purpose — a candidate differing by case or punctuation
|
|
478
|
-
// capital of france" → "What is the capital of France?") is
|
|
479
|
-
// prefix, keeps grounding here, and is unaffected.
|
|
472
|
+
// is exactly what 4,300 competing values must trip. So this is not a new
|
|
473
|
+
// rule and not a new threshold: it is deferring a prefix decision to the
|
|
474
|
+
// tier that owns it (match-project.md, one factored machinery).
|
|
475
|
+
// Byte-strict on purpose — a candidate differing by case or punctuation
|
|
476
|
+
// ("what is the capital of france" → "What is the capital of France?") is
|
|
477
|
+
// NOT a byte prefix, keeps grounding here, and is unaffected.
|
|
480
478
|
const strictPrefix = g !== null &&
|
|
481
479
|
cBytes.length > query.length &&
|
|
482
480
|
indexOf(cBytes, query, 0) === 0;
|
|
@@ -542,15 +540,15 @@ export async function recallByResonance(
|
|
|
542
540
|
}
|
|
543
541
|
}
|
|
544
542
|
|
|
545
|
-
// The refusal/echo decision.
|
|
546
|
-
//
|
|
543
|
+
// The refusal/echo decision. The echo returns a stored form's bytes AS the
|
|
544
|
+
// answer — a near-identity claim about the query — and identity-grade
|
|
547
545
|
// decisions are never made on an estimated score ("approximate scores may
|
|
548
|
-
// rank and propose; they may never decide",
|
|
549
|
-
// overshooting the reach bar echoed a WRONG-entity neighbour
|
|
550
|
-
// Zamunda?" echoed the Armenia fact, observed).
|
|
551
|
-
// anyway to be echoed, so the decision uses their EXACT fold: one river
|
|
552
|
-
// fold of the top hit, measured in the same query-relative,
|
|
553
|
-
//
|
|
546
|
+
// rank and propose; they may never decide", exact-vs-approximate.md): the
|
|
547
|
+
// RaBitQ estimate overshooting the reach bar echoed a WRONG-entity neighbour
|
|
548
|
+
// ("capital of Zamunda?" echoed the Armenia fact, observed). The bytes are
|
|
549
|
+
// read anyway to be echoed, so the decision uses their EXACT fold: one river
|
|
550
|
+
// fold of the top hit, measured in the same query-relative, chance-corrected
|
|
551
|
+
// units as the tier above.
|
|
554
552
|
const topBytes = read(ctx, top.id);
|
|
555
553
|
const exact = topBytes.length > 0
|
|
556
554
|
? cosine(queryGist, gistOf(ctx, topBytes))
|