@hviana/sema 0.8.8 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -184,6 +184,13 @@ export declare class Meter {
184
184
  * route was priced out. Counted where the fact happens (the `!canonBudget`
185
185
  * refusal), not where the probe is called. */
186
186
  canonProbesDenied: number;
187
+ /** Spans the BLOOM claimed and the exact identity then refused — so `probe` used to
188
+ * drop the span AND skip the decider. `findFlatBranch` is bloom-gated on purpose (a
189
+ * miss costs no database read), but a non-null answer means MAYBE; treating it as YES
190
+ * is what made the canon route run only `if (flatProbe === null)`. Counted where the
191
+ * fact happens (a non-null bloom whose `resolveSpan` is null), so the price of letting
192
+ * the canon route see those spans again is a measurement rather than a guess. */
193
+ bloomFalsePositives: number;
187
194
  /** The pipeline's remainder AT THE DECISION POINT, in bytes: what the grounded
188
195
  * answer plus the pre-computed spans left unexplained, after the same W floor
189
196
  * the fuse gate uses. This is the quantity that licenses (or refuses) the
package/dist/src/meter.js CHANGED
@@ -194,6 +194,13 @@ export class Meter {
194
194
  * route was priced out. Counted where the fact happens (the `!canonBudget`
195
195
  * refusal), not where the probe is called. */
196
196
  canonProbesDenied = 0;
197
+ /** Spans the BLOOM claimed and the exact identity then refused — so `probe` used to
198
+ * drop the span AND skip the decider. `findFlatBranch` is bloom-gated on purpose (a
199
+ * miss costs no database read), but a non-null answer means MAYBE; treating it as YES
200
+ * is what made the canon route run only `if (flatProbe === null)`. Counted where the
201
+ * fact happens (a non-null bloom whose `resolveSpan` is null), so the price of letting
202
+ * the canon route see those spans again is a measurement rather than a guess. */
203
+ bloomFalsePositives = 0;
197
204
  /** The pipeline's remainder AT THE DECISION POINT, in bytes: what the grounded
198
205
  * answer plus the pre-computed spans left unexplained, after the same W floor
199
206
  * the fuse gate uses. This is the quantity that licenses (or refuses) the
@@ -518,7 +518,17 @@ function recogniseImpl(ctx, bytes) {
518
518
  // emitted, so a caller can retry a trimmed edge on the miss path only.
519
519
  if (end - start < W)
520
520
  return false;
521
- if (flatProbe(start, end) === null) {
521
+ // The byte-exact route is a BLOOM, so a non-null answer means MAYBE: it can neither
522
+ // decide (that is `resolveSpan`'s job) nor deny (that is `canonAdmits`'). Reading it
523
+ // as a YES made a false positive both drop the span and skip the decider, because the
524
+ // canon route ran only on a null. Measured on the trained corpus: 280 spans where the
525
+ // bloom claimed and the identity refused. A bloom HIT that resolves still emits
526
+ // without touching the canon route, which is what keeps the cheap path cheap.
527
+ const flat = flatProbe(start, end);
528
+ let id = flat === null ? null : resolveSpan(start, end);
529
+ if (id === null) {
530
+ if (flat !== null && ctx.meter)
531
+ ctx.meter.bloomFalsePositives++;
522
532
  if (!canonBudget) {
523
533
  if (ctx.meter)
524
534
  ctx.meter.canonProbesDenied++;
@@ -526,8 +536,8 @@ function recogniseImpl(ctx, bytes) {
526
536
  }
527
537
  if (!canonAdmits(start, end))
528
538
  return false;
539
+ id = resolveSpan(start, end);
529
540
  }
530
- const id = resolveSpan(start, end);
531
541
  if (id === null)
532
542
  return false;
533
543
  emit(start, end, id);
@@ -600,12 +610,21 @@ function recogniseImpl(ctx, bytes) {
600
610
  // keeps this off the quadratic path the budget note above describes (that
601
611
  // one had no span bound at all).
602
612
  {
603
- // The span bound is W^2, the chain's own limit, PLUS the slack the endpoint set already grants: every endpoint
604
- // sits within `radius` of a cut, so a pair that names one form may straddle cuts and still be a single form's
605
- // span. Measured on the composite fixture: W=4 (reach 16), radius 8, and the useful [7,32) is 25 bytes with its
606
- // edges 2 bytes from cuts 5 and 30 — already IN `ordered`, and excluded only by the upper bound. Both terms are
613
+ // The span bound is the chain's own limit PLUS the slack the endpoint set already grants: every endpoint sits
614
+ // within `radius` of a cut, so a pair that names one form may straddle cuts and still be a single form's span.
615
+ // Measured on the composite fixture: W=4 (reach 16), radius 8, and the useful [7,32) is 25 bytes with its edges
616
+ // 2 bytes from cuts 5 and 30 — already IN `ordered`, and excluded only by the upper bound. Both terms are
607
617
  // derived (W and the seat count); no new constant enters.
608
- const reach = chainReach(W) + 2 * radius;
618
+ //
619
+ // The chain term is W^2 * W = W^4, for a form embedded in the MIDDLE of a longer query. Measured on the
620
+ // trained corpus, the SAME real contexts in three positions: opening 12/17, MIDDLE 0/17, end 12/17. The two
621
+ // edge scans reach only prefixes and suffixes, so the interior pass is the only tier that could name it, and
622
+ // at W^2 it could not. WHAT KEEPS THIS LINEAR IS THE BOUND BEING A CONSTANT — each endpoint pairs only with
623
+ // the partners inside a fixed window, so the work stays O(n). Measured against a cut-pair enumeration with
624
+ // ±W trims (an earlier, 43-line attempt): that one reached the same 136 B at 70 888 probes, this one at
625
+ // 84 169 (x1.19 more) — and BOTH pass test/14, which gates the CLASS (linear) and not the constant. The
626
+ // simpler form won on that measurement.
627
+ const reach = chainReach(W) * W * W + 2 * radius;
609
628
  if (ctx.meter)
610
629
  ctx.meter.recogniseInteriorGaps += ordered.length;
611
630
  for (const end of ordered) {
package/jsr.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "$schema": "https://jsr.io/schema/config-file.v1.json",
3
3
  "name": "@hviana/sema",
4
- "version": "0.8.8",
4
+ "version": "0.9.0",
5
5
  "exports": "./src/index.ts"
6
6
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@hviana/sema",
3
- "version": "0.8.8",
3
+ "version": "0.9.0",
4
4
  "description": "Sema: a non-parametric, instance-based reasoning system.",
5
5
  "repository": {
6
6
  "type": "git",
package/src/meter.ts CHANGED
@@ -252,6 +252,13 @@ export class Meter {
252
252
  * route was priced out. Counted where the fact happens (the `!canonBudget`
253
253
  * refusal), not where the probe is called. */
254
254
  canonProbesDenied = 0;
255
+ /** Spans the BLOOM claimed and the exact identity then refused — so `probe` used to
256
+ * drop the span AND skip the decider. `findFlatBranch` is bloom-gated on purpose (a
257
+ * miss costs no database read), but a non-null answer means MAYBE; treating it as YES
258
+ * is what made the canon route run only `if (flatProbe === null)`. Counted where the
259
+ * fact happens (a non-null bloom whose `resolveSpan` is null), so the price of letting
260
+ * the canon route see those spans again is a measurement rather than a guess. */
261
+ bloomFalsePositives = 0;
255
262
  /** The pipeline's remainder AT THE DECISION POINT, in bytes: what the grounded
256
263
  * answer plus the pre-computed spans left unexplained, after the same W floor
257
264
  * the fuse gate uses. This is the quantity that licenses (or refuses) the
@@ -534,14 +534,23 @@ function recogniseImpl(ctx: MindContext, bytes: Uint8Array): Recognition {
534
534
  // pass below spends the same budget on those pairs. Returns whether it
535
535
  // emitted, so a caller can retry a trimmed edge on the miss path only.
536
536
  if (end - start < W) return false;
537
- if (flatProbe(start, end) === null) {
537
+ // The byte-exact route is a BLOOM, so a non-null answer means MAYBE: it can neither
538
+ // decide (that is `resolveSpan`'s job) nor deny (that is `canonAdmits`'). Reading it
539
+ // as a YES made a false positive both drop the span and skip the decider, because the
540
+ // canon route ran only on a null. Measured on the trained corpus: 280 spans where the
541
+ // bloom claimed and the identity refused. A bloom HIT that resolves still emits
542
+ // without touching the canon route, which is what keeps the cheap path cheap.
543
+ const flat = flatProbe(start, end);
544
+ let id = flat === null ? null : resolveSpan(start, end);
545
+ if (id === null) {
546
+ if (flat !== null && ctx.meter) ctx.meter.bloomFalsePositives++;
538
547
  if (!canonBudget) {
539
548
  if (ctx.meter) ctx.meter.canonProbesDenied++;
540
549
  return false;
541
550
  }
542
551
  if (!canonAdmits(start, end)) return false;
552
+ id = resolveSpan(start, end);
543
553
  }
544
- const id = resolveSpan(start, end);
545
554
  if (id === null) return false;
546
555
  emit(start, end, id);
547
556
  return true;
@@ -610,12 +619,21 @@ function recogniseImpl(ctx: MindContext, bytes: Uint8Array): Recognition {
610
619
  // keeps this off the quadratic path the budget note above describes (that
611
620
  // one had no span bound at all).
612
621
  {
613
- // The span bound is W^2, the chain's own limit, PLUS the slack the endpoint set already grants: every endpoint
614
- // sits within `radius` of a cut, so a pair that names one form may straddle cuts and still be a single form's
615
- // span. Measured on the composite fixture: W=4 (reach 16), radius 8, and the useful [7,32) is 25 bytes with its
616
- // edges 2 bytes from cuts 5 and 30 — already IN `ordered`, and excluded only by the upper bound. Both terms are
622
+ // The span bound is the chain's own limit PLUS the slack the endpoint set already grants: every endpoint sits
623
+ // within `radius` of a cut, so a pair that names one form may straddle cuts and still be a single form's span.
624
+ // Measured on the composite fixture: W=4 (reach 16), radius 8, and the useful [7,32) is 25 bytes with its edges
625
+ // 2 bytes from cuts 5 and 30 — already IN `ordered`, and excluded only by the upper bound. Both terms are
617
626
  // derived (W and the seat count); no new constant enters.
618
- const reach = chainReach(W) + 2 * radius;
627
+ //
628
+ // The chain term is W^2 * W = W^4, for a form embedded in the MIDDLE of a longer query. Measured on the
629
+ // trained corpus, the SAME real contexts in three positions: opening 12/17, MIDDLE 0/17, end 12/17. The two
630
+ // edge scans reach only prefixes and suffixes, so the interior pass is the only tier that could name it, and
631
+ // at W^2 it could not. WHAT KEEPS THIS LINEAR IS THE BOUND BEING A CONSTANT — each endpoint pairs only with
632
+ // the partners inside a fixed window, so the work stays O(n). Measured against a cut-pair enumeration with
633
+ // ±W trims (an earlier, 43-line attempt): that one reached the same 136 B at 70 888 probes, this one at
634
+ // 84 169 (x1.19 more) — and BOTH pass test/14, which gates the CLASS (linear) and not the constant. The
635
+ // simpler form won on that measurement.
636
+ const reach = chainReach(W) * W * W + 2 * radius;
619
637
  if (ctx.meter) ctx.meter.recogniseInteriorGaps += ordered.length;
620
638
  for (const end of ordered) {
621
639
  for (const start of ordered) {
@@ -0,0 +1,101 @@
1
+ // 186-embedded-middle.test.mjs — the CUT-PAIR probe of `recognise`: a stored form in the MIDDLE of a query.
2
+ //
3
+ // THE CAPABILITY. A stored form embedded in the MIDDLE of a longer question is named, and its continuation is
4
+ // answered. No tier reached it before: the two edge scans probe only prefixes and suffixes (`spend(0, prefixes[i])`,
5
+ // `spend(s, bytes.length)`), and the interior pass was capped at `reach` = W^2 + 2*radius. Measured on the trained
6
+ // corpus, the SAME real contexts in three positions: opening 12/17, MIDDLE 0/17, end 12/17.
7
+ //
8
+ // HOW. `startList` already holds the content-defined cuts, and both edges of such a form sit within `radius` of a
9
+ // cut. The probe therefore pairs CUTS — the exact pair first, then a ±W trim — under a constant span bound derived
10
+ // from W (`chainReach(W) * W * W`, never a pinned number). The bound is what keeps it linear: an unbounded cut-pair
11
+ // scan is O(cuts^2) and test/14 rejects it.
12
+ //
13
+ // WHAT THIS FILE PINS: the recognition and the ANSWER at all three positions, that an unstored form is still refused
14
+ // (the negative control that keeps the middle assertions from passing for the wrong reason), and the determinism of
15
+ // the reading. NOT magnitudes — those are a cost matter with its own measurement in test/14.
16
+ import test from "node:test";
17
+ import assert from "node:assert/strict";
18
+ import { Mind } from "../dist/src/index.js";
19
+ import { SQliteStore } from "../dist/src/store-sqlite.js";
20
+ import { recognise } from "../dist/src/mind/recognition.js";
21
+
22
+ const dec = new TextDecoder();
23
+ // 64 B, so the form is TWICE `reach` (W^2 + 2*radius) — the dead zone the cut-pair probe exists to close.
24
+ const FORM = "the painter was born in Verano and the river runs through Verano";
25
+ const CONTINUATION = "that region is Calenta";
26
+ const OTHER = "the harbour master";
27
+ const OTHER_CONTINUATION = "the harbour is at Kestrel";
28
+
29
+ async function mind() {
30
+ const m = new Mind({
31
+ seed: 7,
32
+ store: new SQliteStore({ path: ":memory:" }),
33
+ profile: true,
34
+ });
35
+ await m.ingest([
36
+ [FORM, CONTINUATION],
37
+ [OTHER, OTHER_CONTINUATION],
38
+ ]);
39
+ return m;
40
+ }
41
+
42
+ function named(m, query, form) {
43
+ const bytes = new TextEncoder().encode(query);
44
+ return recognise(m, bytes).sites.some((s) =>
45
+ dec.decode(bytes.subarray(s.start, s.end)) === form
46
+ );
47
+ }
48
+
49
+ test("recognise(): a stored form EMBEDDED IN THE MIDDLE of a longer query is named", async () => {
50
+ const m = await mind();
51
+ const query = `I ask: ${OTHER}, ${FORM}, and the note is filed.`;
52
+ assert.equal(named(m, query, FORM), true);
53
+ });
54
+
55
+ test("the middle form is ANSWERED with its own continuation, not the surrounding fact's", async () => {
56
+ const m = await mind();
57
+ const answer = String(
58
+ await m.respondText(
59
+ `I ask: ${OTHER}, ${FORM}, and the note is filed.`,
60
+ () => {},
61
+ ),
62
+ );
63
+ assert.match(answer, /Calenta/);
64
+ });
65
+
66
+ test("the same form at the OPENING and at the END is still named — the edge tiers did not regress", async () => {
67
+ const m = await mind();
68
+ assert.equal(named(m, `${FORM}, and the note is filed.`, FORM), true);
69
+ assert.equal(named(m, `I ask: ${OTHER}, and ${FORM}.`, FORM), true);
70
+ });
71
+
72
+ test("an UNSTORED form is still refused in the middle — the assertion above is not vacuous", async () => {
73
+ const m = await mind();
74
+ // Same LENGTH by construction (one word swapped), and never deposited: the middle assertions must not
75
+ // pass merely because some longer span happens to cover the region.
76
+ const unstored = FORM.replace("painter", "sculpto");
77
+ assert.equal(
78
+ unstored.length,
79
+ FORM.length,
80
+ "the control must be the same size as the form",
81
+ );
82
+ assert.notEqual(unstored, FORM);
83
+ assert.equal(
84
+ named(m, `I ask: ${OTHER}, ${unstored}, and the note is filed.`, unstored),
85
+ false,
86
+ );
87
+ });
88
+
89
+ test("the reading is deterministic, and the interior counter is wired", async () => {
90
+ const first = await mind();
91
+ const second = await mind();
92
+ const query = `I ask: ${OTHER}, ${FORM}, and the note is filed.`;
93
+ const a = String(await first.respondText(query, () => {}));
94
+ const b = String(await second.respondText(query, () => {}));
95
+ assert.equal(a, b);
96
+ const counters = first.lastCost?.counters ?? {};
97
+ assert.ok(
98
+ (counters.recogniseInteriorPairs ?? 0) > 0,
99
+ "recogniseInteriorPairs must be wired",
100
+ );
101
+ });