@hviana/sema 0.8.8 → 0.8.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/src/mind/recognition.js +53 -0
- package/jsr.json +1 -1
- package/package.json +1 -1
- package/src/mind/recognition.ts +49 -0
- package/test/186-embedded-middle.test.mjs +101 -0
|
@@ -620,6 +620,59 @@ function recogniseImpl(ctx, bytes) {
|
|
|
620
620
|
spend(start, end);
|
|
621
621
|
}
|
|
622
622
|
}
|
|
623
|
+
// ── THE MIDDLE WAS BLIND: a CUT-PAIR probe, BOUNDED so it stays LINEAR ──────────────
|
|
624
|
+
// A stored form embedded in the MIDDLE of a longer query was named by NO tier: the two
|
|
625
|
+
// edge scans probe only prefixes and suffixes, and the loop above is capped at
|
|
626
|
+
// `reach` = W^2 + 2*radius. Measured on the trained corpus, the SAME real contexts in
|
|
627
|
+
// three positions: opening 12/17, MIDDLE 0/17, end 12/17. Both edges of such a form sit
|
|
628
|
+
// within `radius` of CUTS, so a cut pair plus a small trim names the form exactly
|
|
629
|
+
// (measured: a 64 B form at [27, 91) between cuts 26 and 90, trims +1/+1).
|
|
630
|
+
//
|
|
631
|
+
// THE BOUND IS WHAT KEEPS IT LINEAR. An unbounded cut-pair scan is O(cuts^2) probes, and
|
|
632
|
+
// test/14 rejected exactly that (47 324 ms). Bounded by a CONSTANT span it is
|
|
633
|
+
// O(cuts * const) = O(n), like the loop above — the constant is merely larger. The bound
|
|
634
|
+
// is DERIVED, never pinned: `chainReach(W)` is W^2, "the deepest two-level composite the
|
|
635
|
+
// write side's windows can spell" (canonical.ts), so `chainReach(W) * W * W` is the
|
|
636
|
+
// four-level one; `chainReach(W) * W` is already used in bridge.ts.
|
|
637
|
+
//
|
|
638
|
+
// THE TRIMS ARE THE SAME DISCIPLINE THE EDGE SCANS USE — the suffix scan already probes
|
|
639
|
+
// `spend(s, bytes.length - 1)`, so ±W is a wider version of an existing rule, not a new
|
|
640
|
+
// threshold. Measured price/reach; all four configurations pass test/14, which gates the
|
|
641
|
+
// CLASS (linear) and not the constant:
|
|
642
|
+
// W^3 + ±1 -> 64 B, 8 068 probes (x2.02)
|
|
643
|
+
// W^4 + ±W -> 136 B, 70 888 probes (x6.06) <- this one
|
|
644
|
+
// W^3 + ±W and W^4 + ±1 are DOMINATED: each stops at the other parameter and costs more.
|
|
645
|
+
// The loop above stays, so no candidate that produces a site today is lost. Suite
|
|
646
|
+
// 707/707, and a differential over 24 real corpus questions is byte-identical (0 answers
|
|
647
|
+
// changed, 0 new duplicate sites): the :258 warning that a wider bound can rediscover a
|
|
648
|
+
// smaller subtree's content as a second, overlapping site was read and measured, and it
|
|
649
|
+
// does not materialise here.
|
|
650
|
+
const deepReach = chainReach(W) * W * W;
|
|
651
|
+
for (let ci = 0; ci + 1 < startList.length; ci++) {
|
|
652
|
+
for (let cj = ci + 1; cj < startList.length; cj++) {
|
|
653
|
+
if (startList[cj] - startList[ci] > deepReach + 2 * radius)
|
|
654
|
+
break;
|
|
655
|
+
// The EXACT cut pair first. Measured: with the 81 trims starting at -W the `spend`
|
|
656
|
+
// pool ran dry (canonProbesDenied in the millions) and the canon route was then denied
|
|
657
|
+
// to the candidates that needed it — including forms the byte-exact route SEES
|
|
658
|
+
// (`flatProbe` true) that were still not named. Probing (ci, cj) before any trim puts
|
|
659
|
+
// the common case in front of the famine.
|
|
660
|
+
if (ctx.meter)
|
|
661
|
+
ctx.meter.recogniseInteriorPairs++;
|
|
662
|
+
spend(startList[ci], startList[cj]);
|
|
663
|
+
for (let dl = -W; dl <= W; dl++) {
|
|
664
|
+
for (let dr = -W; dr <= W; dr++) {
|
|
665
|
+
const a = startList[ci] + dl;
|
|
666
|
+
const z = startList[cj] + dr;
|
|
667
|
+
if (a < 0 || z > bytes.length || z - a < W)
|
|
668
|
+
continue;
|
|
669
|
+
if (ctx.meter)
|
|
670
|
+
ctx.meter.recogniseInteriorPairs++;
|
|
671
|
+
spend(a, z);
|
|
672
|
+
}
|
|
673
|
+
}
|
|
674
|
+
}
|
|
675
|
+
}
|
|
623
676
|
}
|
|
624
677
|
}
|
|
625
678
|
}
|
package/jsr.json
CHANGED
package/package.json
CHANGED
package/src/mind/recognition.ts
CHANGED
|
@@ -626,6 +626,55 @@ function recogniseImpl(ctx: MindContext, bytes: Uint8Array): Recognition {
|
|
|
626
626
|
spend(start, end);
|
|
627
627
|
}
|
|
628
628
|
}
|
|
629
|
+
// ── THE MIDDLE WAS BLIND: a CUT-PAIR probe, BOUNDED so it stays LINEAR ──────────────
|
|
630
|
+
// A stored form embedded in the MIDDLE of a longer query was named by NO tier: the two
|
|
631
|
+
// edge scans probe only prefixes and suffixes, and the loop above is capped at
|
|
632
|
+
// `reach` = W^2 + 2*radius. Measured on the trained corpus, the SAME real contexts in
|
|
633
|
+
// three positions: opening 12/17, MIDDLE 0/17, end 12/17. Both edges of such a form sit
|
|
634
|
+
// within `radius` of CUTS, so a cut pair plus a small trim names the form exactly
|
|
635
|
+
// (measured: a 64 B form at [27, 91) between cuts 26 and 90, trims +1/+1).
|
|
636
|
+
//
|
|
637
|
+
// THE BOUND IS WHAT KEEPS IT LINEAR. An unbounded cut-pair scan is O(cuts^2) probes, and
|
|
638
|
+
// test/14 rejected exactly that (47 324 ms). Bounded by a CONSTANT span it is
|
|
639
|
+
// O(cuts * const) = O(n), like the loop above — the constant is merely larger. The bound
|
|
640
|
+
// is DERIVED, never pinned: `chainReach(W)` is W^2, "the deepest two-level composite the
|
|
641
|
+
// write side's windows can spell" (canonical.ts), so `chainReach(W) * W * W` is the
|
|
642
|
+
// four-level one; `chainReach(W) * W` is already used in bridge.ts.
|
|
643
|
+
//
|
|
644
|
+
// THE TRIMS ARE THE SAME DISCIPLINE THE EDGE SCANS USE — the suffix scan already probes
|
|
645
|
+
// `spend(s, bytes.length - 1)`, so ±W is a wider version of an existing rule, not a new
|
|
646
|
+
// threshold. Measured price/reach; all four configurations pass test/14, which gates the
|
|
647
|
+
// CLASS (linear) and not the constant:
|
|
648
|
+
// W^3 + ±1 -> 64 B, 8 068 probes (x2.02)
|
|
649
|
+
// W^4 + ±W -> 136 B, 70 888 probes (x6.06) <- this one
|
|
650
|
+
// W^3 + ±W and W^4 + ±1 are DOMINATED: each stops at the other parameter and costs more.
|
|
651
|
+
// The loop above stays, so no candidate that produces a site today is lost. Suite
|
|
652
|
+
// 707/707, and a differential over 24 real corpus questions is byte-identical (0 answers
|
|
653
|
+
// changed, 0 new duplicate sites): the :258 warning that a wider bound can rediscover a
|
|
654
|
+
// smaller subtree's content as a second, overlapping site was read and measured, and it
|
|
655
|
+
// does not materialise here.
|
|
656
|
+
const deepReach = chainReach(W) * W * W;
|
|
657
|
+
for (let ci = 0; ci + 1 < startList.length; ci++) {
|
|
658
|
+
for (let cj = ci + 1; cj < startList.length; cj++) {
|
|
659
|
+
if (startList[cj] - startList[ci] > deepReach + 2 * radius) break;
|
|
660
|
+
// The EXACT cut pair first. Measured: with the 81 trims starting at -W the `spend`
|
|
661
|
+
// pool ran dry (canonProbesDenied in the millions) and the canon route was then denied
|
|
662
|
+
// to the candidates that needed it — including forms the byte-exact route SEES
|
|
663
|
+
// (`flatProbe` true) that were still not named. Probing (ci, cj) before any trim puts
|
|
664
|
+
// the common case in front of the famine.
|
|
665
|
+
if (ctx.meter) ctx.meter.recogniseInteriorPairs++;
|
|
666
|
+
spend(startList[ci], startList[cj]);
|
|
667
|
+
for (let dl = -W; dl <= W; dl++) {
|
|
668
|
+
for (let dr = -W; dr <= W; dr++) {
|
|
669
|
+
const a = startList[ci] + dl;
|
|
670
|
+
const z = startList[cj] + dr;
|
|
671
|
+
if (a < 0 || z > bytes.length || z - a < W) continue;
|
|
672
|
+
if (ctx.meter) ctx.meter.recogniseInteriorPairs++;
|
|
673
|
+
spend(a, z);
|
|
674
|
+
}
|
|
675
|
+
}
|
|
676
|
+
}
|
|
677
|
+
}
|
|
629
678
|
}
|
|
630
679
|
}
|
|
631
680
|
}
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
// 186-embedded-middle.test.mjs — the CUT-PAIR probe of `recognise`: a stored form in the MIDDLE of a query.
|
|
2
|
+
//
|
|
3
|
+
// THE CAPABILITY. A stored form embedded in the MIDDLE of a longer question is named, and its continuation is
|
|
4
|
+
// answered. No tier reached it before: the two edge scans probe only prefixes and suffixes (`spend(0, prefixes[i])`,
|
|
5
|
+
// `spend(s, bytes.length)`), and the interior pass was capped at `reach` = W^2 + 2*radius. Measured on the trained
|
|
6
|
+
// corpus, the SAME real contexts in three positions: opening 12/17, MIDDLE 0/17, end 12/17.
|
|
7
|
+
//
|
|
8
|
+
// HOW. `startList` already holds the content-defined cuts, and both edges of such a form sit within `radius` of a
|
|
9
|
+
// cut. The probe therefore pairs CUTS — the exact pair first, then a ±W trim — under a constant span bound derived
|
|
10
|
+
// from W (`chainReach(W) * W * W`, never a pinned number). The bound is what keeps it linear: an unbounded cut-pair
|
|
11
|
+
// scan is O(cuts^2) and test/14 rejects it.
|
|
12
|
+
//
|
|
13
|
+
// WHAT THIS FILE PINS: the recognition and the ANSWER at all three positions, that an unstored form is still refused
|
|
14
|
+
// (the negative control that keeps the middle assertions from passing for the wrong reason), and the determinism of
|
|
15
|
+
// the reading. NOT magnitudes — those are a cost matter with its own measurement in test/14.
|
|
16
|
+
import test from "node:test";
|
|
17
|
+
import assert from "node:assert/strict";
|
|
18
|
+
import { Mind } from "../dist/src/index.js";
|
|
19
|
+
import { SQliteStore } from "../dist/src/store-sqlite.js";
|
|
20
|
+
import { recognise } from "../dist/src/mind/recognition.js";
|
|
21
|
+
|
|
22
|
+
const dec = new TextDecoder();
|
|
23
|
+
// 64 B, so the form is TWICE `reach` (W^2 + 2*radius) — the dead zone the cut-pair probe exists to close.
|
|
24
|
+
const FORM = "the painter was born in Verano and the river runs through Verano";
|
|
25
|
+
const CONTINUATION = "that region is Calenta";
|
|
26
|
+
const OTHER = "the harbour master";
|
|
27
|
+
const OTHER_CONTINUATION = "the harbour is at Kestrel";
|
|
28
|
+
|
|
29
|
+
async function mind() {
|
|
30
|
+
const m = new Mind({
|
|
31
|
+
seed: 7,
|
|
32
|
+
store: new SQliteStore({ path: ":memory:" }),
|
|
33
|
+
profile: true,
|
|
34
|
+
});
|
|
35
|
+
await m.ingest([
|
|
36
|
+
[FORM, CONTINUATION],
|
|
37
|
+
[OTHER, OTHER_CONTINUATION],
|
|
38
|
+
]);
|
|
39
|
+
return m;
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
function named(m, query, form) {
|
|
43
|
+
const bytes = new TextEncoder().encode(query);
|
|
44
|
+
return recognise(m, bytes).sites.some((s) =>
|
|
45
|
+
dec.decode(bytes.subarray(s.start, s.end)) === form
|
|
46
|
+
);
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
test("recognise(): a stored form EMBEDDED IN THE MIDDLE of a longer query is named", async () => {
|
|
50
|
+
const m = await mind();
|
|
51
|
+
const query = `I ask: ${OTHER}, ${FORM}, and the note is filed.`;
|
|
52
|
+
assert.equal(named(m, query, FORM), true);
|
|
53
|
+
});
|
|
54
|
+
|
|
55
|
+
test("the middle form is ANSWERED with its own continuation, not the surrounding fact's", async () => {
|
|
56
|
+
const m = await mind();
|
|
57
|
+
const answer = String(
|
|
58
|
+
await m.respondText(
|
|
59
|
+
`I ask: ${OTHER}, ${FORM}, and the note is filed.`,
|
|
60
|
+
() => {},
|
|
61
|
+
),
|
|
62
|
+
);
|
|
63
|
+
assert.match(answer, /Calenta/);
|
|
64
|
+
});
|
|
65
|
+
|
|
66
|
+
test("the same form at the OPENING and at the END is still named — the edge tiers did not regress", async () => {
|
|
67
|
+
const m = await mind();
|
|
68
|
+
assert.equal(named(m, `${FORM}, and the note is filed.`, FORM), true);
|
|
69
|
+
assert.equal(named(m, `I ask: ${OTHER}, and ${FORM}.`, FORM), true);
|
|
70
|
+
});
|
|
71
|
+
|
|
72
|
+
test("an UNSTORED form is still refused in the middle — the assertion above is not vacuous", async () => {
|
|
73
|
+
const m = await mind();
|
|
74
|
+
// Same LENGTH by construction (one word swapped), and never deposited: the middle assertions must not
|
|
75
|
+
// pass merely because some longer span happens to cover the region.
|
|
76
|
+
const unstored = FORM.replace("painter", "sculpto");
|
|
77
|
+
assert.equal(
|
|
78
|
+
unstored.length,
|
|
79
|
+
FORM.length,
|
|
80
|
+
"the control must be the same size as the form",
|
|
81
|
+
);
|
|
82
|
+
assert.notEqual(unstored, FORM);
|
|
83
|
+
assert.equal(
|
|
84
|
+
named(m, `I ask: ${OTHER}, ${unstored}, and the note is filed.`, unstored),
|
|
85
|
+
false,
|
|
86
|
+
);
|
|
87
|
+
});
|
|
88
|
+
|
|
89
|
+
test("the reading is deterministic, and the interior counter is wired", async () => {
|
|
90
|
+
const first = await mind();
|
|
91
|
+
const second = await mind();
|
|
92
|
+
const query = `I ask: ${OTHER}, ${FORM}, and the note is filed.`;
|
|
93
|
+
const a = String(await first.respondText(query, () => {}));
|
|
94
|
+
const b = String(await second.respondText(query, () => {}));
|
|
95
|
+
assert.equal(a, b);
|
|
96
|
+
const counters = first.lastCost?.counters ?? {};
|
|
97
|
+
assert.ok(
|
|
98
|
+
(counters.recogniseInteriorPairs ?? 0) > 0,
|
|
99
|
+
"recogniseInteriorPairs must be wired",
|
|
100
|
+
);
|
|
101
|
+
});
|