@hviana/sema 0.8.8 → 0.8.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -620,6 +620,59 @@ function recogniseImpl(ctx, bytes) {
620
620
  spend(start, end);
621
621
  }
622
622
  }
623
+ // ── THE MIDDLE WAS BLIND: a CUT-PAIR probe, BOUNDED so it stays LINEAR ──────────────
624
+ // A stored form embedded in the MIDDLE of a longer query was named by NO tier: the two
625
+ // edge scans probe only prefixes and suffixes, and the loop above is capped at
626
+ // `reach` = W^2 + 2*radius. Measured on the trained corpus, the SAME real contexts in
627
+ // three positions: opening 12/17, MIDDLE 0/17, end 12/17. Both edges of such a form sit
628
+ // within `radius` of CUTS, so a cut pair plus a small trim names the form exactly
629
+ // (measured: a 64 B form at [27, 91) between cuts 26 and 90, trims +1/+1).
630
+ //
631
+ // THE BOUND IS WHAT KEEPS IT LINEAR. An unbounded cut-pair scan is O(cuts^2) probes, and
632
+ // test/14 rejected exactly that (47 324 ms). Bounded by a CONSTANT span it is
633
+ // O(cuts * const) = O(n), like the loop above — the constant is merely larger. The bound
634
+ // is DERIVED, never pinned: `chainReach(W)` is W^2, "the deepest two-level composite the
635
+ // write side's windows can spell" (canonical.ts), so `chainReach(W) * W * W` is the
636
+ // four-level one; `chainReach(W) * W` is already used in bridge.ts.
637
+ //
638
+ // THE TRIMS ARE THE SAME DISCIPLINE THE EDGE SCANS USE — the suffix scan already probes
639
+ // `spend(s, bytes.length - 1)`, so ±W is a wider version of an existing rule, not a new
640
+ // threshold. Measured price/reach; all four configurations pass test/14, which gates the
641
+ // CLASS (linear) and not the constant:
642
+ // W^3 + ±1 -> 64 B, 8 068 probes (x2.02)
643
+ // W^4 + ±W -> 136 B, 70 888 probes (x6.06) <- this one
644
+ // W^3 + ±W and W^4 + ±1 are DOMINATED: each stops at the other parameter and costs more.
645
+ // The loop above stays, so no candidate that produces a site today is lost. Suite
646
+ // 707/707, and a differential over 24 real corpus questions is byte-identical (0 answers
647
+ // changed, 0 new duplicate sites): the :258 warning that a wider bound can rediscover a
648
+ // smaller subtree's content as a second, overlapping site was read and measured, and it
649
+ // does not materialise here.
650
+ const deepReach = chainReach(W) * W * W;
651
+ for (let ci = 0; ci + 1 < startList.length; ci++) {
652
+ for (let cj = ci + 1; cj < startList.length; cj++) {
653
+ if (startList[cj] - startList[ci] > deepReach + 2 * radius)
654
+ break;
655
+ // The EXACT cut pair first. Measured: with the 81 trims starting at -W the `spend`
656
+ // pool ran dry (canonProbesDenied in the millions) and the canon route was then denied
657
+ // to the candidates that needed it — including forms the byte-exact route SEES
658
+ // (`flatProbe` true) that were still not named. Probing (ci, cj) before any trim puts
659
+ // the common case in front of the famine.
660
+ if (ctx.meter)
661
+ ctx.meter.recogniseInteriorPairs++;
662
+ spend(startList[ci], startList[cj]);
663
+ for (let dl = -W; dl <= W; dl++) {
664
+ for (let dr = -W; dr <= W; dr++) {
665
+ const a = startList[ci] + dl;
666
+ const z = startList[cj] + dr;
667
+ if (a < 0 || z > bytes.length || z - a < W)
668
+ continue;
669
+ if (ctx.meter)
670
+ ctx.meter.recogniseInteriorPairs++;
671
+ spend(a, z);
672
+ }
673
+ }
674
+ }
675
+ }
623
676
  }
624
677
  }
625
678
  }
package/jsr.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "$schema": "https://jsr.io/schema/config-file.v1.json",
3
3
  "name": "@hviana/sema",
4
- "version": "0.8.8",
4
+ "version": "0.8.9",
5
5
  "exports": "./src/index.ts"
6
6
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@hviana/sema",
3
- "version": "0.8.8",
3
+ "version": "0.8.9",
4
4
  "description": "Sema: a non-parametric, instance-based reasoning system.",
5
5
  "repository": {
6
6
  "type": "git",
@@ -626,6 +626,55 @@ function recogniseImpl(ctx: MindContext, bytes: Uint8Array): Recognition {
626
626
  spend(start, end);
627
627
  }
628
628
  }
629
+ // ── THE MIDDLE WAS BLIND: a CUT-PAIR probe, BOUNDED so it stays LINEAR ──────────────
630
+ // A stored form embedded in the MIDDLE of a longer query was named by NO tier: the two
631
+ // edge scans probe only prefixes and suffixes, and the loop above is capped at
632
+ // `reach` = W^2 + 2*radius. Measured on the trained corpus, the SAME real contexts in
633
+ // three positions: opening 12/17, MIDDLE 0/17, end 12/17. Both edges of such a form sit
634
+ // within `radius` of CUTS, so a cut pair plus a small trim names the form exactly
635
+ // (measured: a 64 B form at [27, 91) between cuts 26 and 90, trims +1/+1).
636
+ //
637
+ // THE BOUND IS WHAT KEEPS IT LINEAR. An unbounded cut-pair scan is O(cuts^2) probes, and
638
+ // test/14 rejected exactly that (47 324 ms). Bounded by a CONSTANT span it is
639
+ // O(cuts * const) = O(n), like the loop above — the constant is merely larger. The bound
640
+ // is DERIVED, never pinned: `chainReach(W)` is W^2, "the deepest two-level composite the
641
+ // write side's windows can spell" (canonical.ts), so `chainReach(W) * W * W` is the
642
+ // four-level one; `chainReach(W) * W` is already used in bridge.ts.
643
+ //
644
+ // THE TRIMS ARE THE SAME DISCIPLINE THE EDGE SCANS USE — the suffix scan already probes
645
+ // `spend(s, bytes.length - 1)`, so ±W is a wider version of an existing rule, not a new
646
+ // threshold. Measured price/reach; all four configurations pass test/14, which gates the
647
+ // CLASS (linear) and not the constant:
648
+ // W^3 + ±1 -> 64 B, 8 068 probes (x2.02)
649
+ // W^4 + ±W -> 136 B, 70 888 probes (x6.06) <- this one
650
+ // W^3 + ±W and W^4 + ±1 are DOMINATED: each stops at the other parameter and costs more.
651
+ // The loop above stays, so no candidate that produces a site today is lost. Suite
652
+ // 707/707, and a differential over 24 real corpus questions is byte-identical (0 answers
653
+ // changed, 0 new duplicate sites): the :258 warning that a wider bound can rediscover a
654
+ // smaller subtree's content as a second, overlapping site was read and measured, and it
655
+ // does not materialise here.
656
+ const deepReach = chainReach(W) * W * W;
657
+ for (let ci = 0; ci + 1 < startList.length; ci++) {
658
+ for (let cj = ci + 1; cj < startList.length; cj++) {
659
+ if (startList[cj] - startList[ci] > deepReach + 2 * radius) break;
660
+ // The EXACT cut pair first. Measured: with the 81 trims starting at -W the `spend`
661
+ // pool ran dry (canonProbesDenied in the millions) and the canon route was then denied
662
+ // to the candidates that needed it — including forms the byte-exact route SEES
663
+ // (`flatProbe` true) that were still not named. Probing (ci, cj) before any trim puts
664
+ // the common case in front of the famine.
665
+ if (ctx.meter) ctx.meter.recogniseInteriorPairs++;
666
+ spend(startList[ci], startList[cj]);
667
+ for (let dl = -W; dl <= W; dl++) {
668
+ for (let dr = -W; dr <= W; dr++) {
669
+ const a = startList[ci] + dl;
670
+ const z = startList[cj] + dr;
671
+ if (a < 0 || z > bytes.length || z - a < W) continue;
672
+ if (ctx.meter) ctx.meter.recogniseInteriorPairs++;
673
+ spend(a, z);
674
+ }
675
+ }
676
+ }
677
+ }
629
678
  }
630
679
  }
631
680
  }
@@ -0,0 +1,101 @@
1
+ // 186-embedded-middle.test.mjs — the CUT-PAIR probe of `recognise`: a stored form in the MIDDLE of a query.
2
+ //
3
+ // THE CAPABILITY. A stored form embedded in the MIDDLE of a longer question is named, and its continuation is
4
+ // answered. No tier reached it before: the two edge scans probe only prefixes and suffixes (`spend(0, prefixes[i])`,
5
+ // `spend(s, bytes.length)`), and the interior pass was capped at `reach` = W^2 + 2*radius. Measured on the trained
6
+ // corpus, the SAME real contexts in three positions: opening 12/17, MIDDLE 0/17, end 12/17.
7
+ //
8
+ // HOW. `startList` already holds the content-defined cuts, and both edges of such a form sit within `radius` of a
9
+ // cut. The probe therefore pairs CUTS — the exact pair first, then a ±W trim — under a constant span bound derived
10
+ // from W (`chainReach(W) * W * W`, never a pinned number). The bound is what keeps it linear: an unbounded cut-pair
11
+ // scan is O(cuts^2) and test/14 rejects it.
12
+ //
13
+ // WHAT THIS FILE PINS: the recognition and the ANSWER at all three positions, that an unstored form is still refused
14
+ // (the negative control that keeps the middle assertions from passing for the wrong reason), and the determinism of
15
+ // the reading. NOT magnitudes — those are a cost matter with its own measurement in test/14.
16
+ import test from "node:test";
17
+ import assert from "node:assert/strict";
18
+ import { Mind } from "../dist/src/index.js";
19
+ import { SQliteStore } from "../dist/src/store-sqlite.js";
20
+ import { recognise } from "../dist/src/mind/recognition.js";
21
+
22
+ const dec = new TextDecoder();
23
+ // 64 B, so the form is TWICE `reach` (W^2 + 2*radius) — the dead zone the cut-pair probe exists to close.
24
+ const FORM = "the painter was born in Verano and the river runs through Verano";
25
+ const CONTINUATION = "that region is Calenta";
26
+ const OTHER = "the harbour master";
27
+ const OTHER_CONTINUATION = "the harbour is at Kestrel";
28
+
29
+ async function mind() {
30
+ const m = new Mind({
31
+ seed: 7,
32
+ store: new SQliteStore({ path: ":memory:" }),
33
+ profile: true,
34
+ });
35
+ await m.ingest([
36
+ [FORM, CONTINUATION],
37
+ [OTHER, OTHER_CONTINUATION],
38
+ ]);
39
+ return m;
40
+ }
41
+
42
+ function named(m, query, form) {
43
+ const bytes = new TextEncoder().encode(query);
44
+ return recognise(m, bytes).sites.some((s) =>
45
+ dec.decode(bytes.subarray(s.start, s.end)) === form
46
+ );
47
+ }
48
+
49
+ test("recognise(): a stored form EMBEDDED IN THE MIDDLE of a longer query is named", async () => {
50
+ const m = await mind();
51
+ const query = `I ask: ${OTHER}, ${FORM}, and the note is filed.`;
52
+ assert.equal(named(m, query, FORM), true);
53
+ });
54
+
55
+ test("the middle form is ANSWERED with its own continuation, not the surrounding fact's", async () => {
56
+ const m = await mind();
57
+ const answer = String(
58
+ await m.respondText(
59
+ `I ask: ${OTHER}, ${FORM}, and the note is filed.`,
60
+ () => {},
61
+ ),
62
+ );
63
+ assert.match(answer, /Calenta/);
64
+ });
65
+
66
+ test("the same form at the OPENING and at the END is still named — the edge tiers did not regress", async () => {
67
+ const m = await mind();
68
+ assert.equal(named(m, `${FORM}, and the note is filed.`, FORM), true);
69
+ assert.equal(named(m, `I ask: ${OTHER}, and ${FORM}.`, FORM), true);
70
+ });
71
+
72
+ test("an UNSTORED form is still refused in the middle — the assertion above is not vacuous", async () => {
73
+ const m = await mind();
74
+ // Same LENGTH by construction (one word swapped), and never deposited: the middle assertions must not
75
+ // pass merely because some longer span happens to cover the region.
76
+ const unstored = FORM.replace("painter", "sculpto");
77
+ assert.equal(
78
+ unstored.length,
79
+ FORM.length,
80
+ "the control must be the same size as the form",
81
+ );
82
+ assert.notEqual(unstored, FORM);
83
+ assert.equal(
84
+ named(m, `I ask: ${OTHER}, ${unstored}, and the note is filed.`, unstored),
85
+ false,
86
+ );
87
+ });
88
+
89
+ test("the reading is deterministic, and the interior counter is wired", async () => {
90
+ const first = await mind();
91
+ const second = await mind();
92
+ const query = `I ask: ${OTHER}, ${FORM}, and the note is filed.`;
93
+ const a = String(await first.respondText(query, () => {}));
94
+ const b = String(await second.respondText(query, () => {}));
95
+ assert.equal(a, b);
96
+ const counters = first.lastCost?.counters ?? {};
97
+ assert.ok(
98
+ (counters.recogniseInteriorPairs ?? 0) > 0,
99
+ "recogniseInteriorPairs must be wired",
100
+ );
101
+ });